CodeIsAbstract commited on
Commit
0149061
·
verified ·
1 Parent(s): a750c95

Training in progress, step 24000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f4c37dc9cd45148160db4b80fa4db4da68c9b444487ae1c9fdf7890466cd3b11
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:91939e0b72ed547f5c93280b369185295f37271995fd46a6edb271145c07605c
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:77dfb0dfb6c162187d26607f9585f8626cc40ff65b297436f2124b2fe6a83c10
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c4519573d36d90ba131c7875410ba23ca2fd486818679adec360ef2a2dd49f18
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a7e057f5b72466889872d74430f3ee9303ae077763a80c9351cb7f41490bb946
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68a1e729745879daefb80abbed0a2e5c919e4730374ec35c85a8be9176df6bb5
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f253e447661d715656ec6271ae4c4ea6bd1781e5f900ea4b2d9b76d2fc849b12
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d44e1b79b7ca3c1d7f485201d49324d3b81b5df8a48270324b94dd9cdfe85a81
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.18181818181818182,
6
  "eval_steps": 1000,
7
- "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1568,6 +1568,318 @@
1568
  "eval_samples_per_second": 84.397,
1569
  "eval_steps_per_second": 21.099,
1570
  "step": 20000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1571
  }
1572
  ],
1573
  "logging_steps": 100,
@@ -1587,7 +1899,7 @@
1587
  "attributes": {}
1588
  }
1589
  },
1590
- "total_flos": 4.9829161992192e+17,
1591
  "train_batch_size": 22,
1592
  "trial_name": null,
1593
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.21818181818181817,
6
  "eval_steps": 1000,
7
+ "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1568
  "eval_samples_per_second": 84.397,
1569
  "eval_steps_per_second": 21.099,
1570
  "step": 20000
1571
+ },
1572
+ {
1573
+ "epoch": 0.18272727272727274,
1574
+ "grad_norm": 0.1414630264043808,
1575
+ "learning_rate": 0.000920810102137561,
1576
+ "loss": 2.8807022094726564,
1577
+ "step": 20100
1578
+ },
1579
+ {
1580
+ "epoch": 0.18363636363636363,
1581
+ "grad_norm": 0.1559121161699295,
1582
+ "learning_rate": 0.0009200361112663695,
1583
+ "loss": 2.845583190917969,
1584
+ "step": 20200
1585
+ },
1586
+ {
1587
+ "epoch": 0.18454545454545454,
1588
+ "grad_norm": 0.13616813719272614,
1589
+ "learning_rate": 0.0009192586849267957,
1590
+ "loss": 2.8783746337890626,
1591
+ "step": 20300
1592
+ },
1593
+ {
1594
+ "epoch": 0.18545454545454546,
1595
+ "grad_norm": 0.15556134283542633,
1596
+ "learning_rate": 0.0009184778294773968,
1597
+ "loss": 2.84674072265625,
1598
+ "step": 20400
1599
+ },
1600
+ {
1601
+ "epoch": 0.18636363636363637,
1602
+ "grad_norm": 0.1401907503604889,
1603
+ "learning_rate": 0.0009176935513047762,
1604
+ "loss": 2.8818963623046874,
1605
+ "step": 20500
1606
+ },
1607
+ {
1608
+ "epoch": 0.18727272727272729,
1609
+ "grad_norm": 0.1431131362915039,
1610
+ "learning_rate": 0.0009169058568235323,
1611
+ "loss": 2.8781854248046876,
1612
+ "step": 20600
1613
+ },
1614
+ {
1615
+ "epoch": 0.18818181818181817,
1616
+ "grad_norm": 0.17515692114830017,
1617
+ "learning_rate": 0.0009161147524762053,
1618
+ "loss": 2.85051025390625,
1619
+ "step": 20700
1620
+ },
1621
+ {
1622
+ "epoch": 0.1890909090909091,
1623
+ "grad_norm": 0.1567743867635727,
1624
+ "learning_rate": 0.0009153202447332243,
1625
+ "loss": 2.8653643798828123,
1626
+ "step": 20800
1627
+ },
1628
+ {
1629
+ "epoch": 0.19,
1630
+ "grad_norm": 0.14393377304077148,
1631
+ "learning_rate": 0.000914522340092855,
1632
+ "loss": 2.881787109375,
1633
+ "step": 20900
1634
+ },
1635
+ {
1636
+ "epoch": 0.19090909090909092,
1637
+ "grad_norm": 0.1467711478471756,
1638
+ "learning_rate": 0.0009137210450811462,
1639
+ "loss": 2.9003524780273438,
1640
+ "step": 21000
1641
+ },
1642
+ {
1643
+ "epoch": 0.19090909090909092,
1644
+ "eval_loss": 3.210813283920288,
1645
+ "eval_runtime": 6.8446,
1646
+ "eval_samples_per_second": 84.154,
1647
+ "eval_steps_per_second": 21.038,
1648
+ "step": 21000
1649
+ },
1650
+ {
1651
+ "epoch": 0.1918181818181818,
1652
+ "grad_norm": 0.14093157649040222,
1653
+ "learning_rate": 0.0009129163662518766,
1654
+ "loss": 2.834397277832031,
1655
+ "step": 21100
1656
+ },
1657
+ {
1658
+ "epoch": 0.19272727272727272,
1659
+ "grad_norm": 0.14963804185390472,
1660
+ "learning_rate": 0.000912108310186501,
1661
+ "loss": 2.869900817871094,
1662
+ "step": 21200
1663
+ },
1664
+ {
1665
+ "epoch": 0.19363636363636363,
1666
+ "grad_norm": 0.13998101651668549,
1667
+ "learning_rate": 0.0009112968834940965,
1668
+ "loss": 2.8689230346679686,
1669
+ "step": 21300
1670
+ },
1671
+ {
1672
+ "epoch": 0.19454545454545455,
1673
+ "grad_norm": 0.13820187747478485,
1674
+ "learning_rate": 0.0009104820928113083,
1675
+ "loss": 2.840967102050781,
1676
+ "step": 21400
1677
+ },
1678
+ {
1679
+ "epoch": 0.19545454545454546,
1680
+ "grad_norm": 0.1391438990831375,
1681
+ "learning_rate": 0.0009096639448022963,
1682
+ "loss": 2.839700927734375,
1683
+ "step": 21500
1684
+ },
1685
+ {
1686
+ "epoch": 0.19636363636363635,
1687
+ "grad_norm": 0.15131688117980957,
1688
+ "learning_rate": 0.0009088424461586793,
1689
+ "loss": 2.8546661376953124,
1690
+ "step": 21600
1691
+ },
1692
+ {
1693
+ "epoch": 0.19727272727272727,
1694
+ "grad_norm": 0.14853839576244354,
1695
+ "learning_rate": 0.000908017603599481,
1696
+ "loss": 2.8781912231445315,
1697
+ "step": 21700
1698
+ },
1699
+ {
1700
+ "epoch": 0.19818181818181818,
1701
+ "grad_norm": 0.16618646681308746,
1702
+ "learning_rate": 0.0009071894238710749,
1703
+ "loss": 2.8425857543945314,
1704
+ "step": 21800
1705
+ },
1706
+ {
1707
+ "epoch": 0.1990909090909091,
1708
+ "grad_norm": 0.13768044114112854,
1709
+ "learning_rate": 0.0009063579137471296,
1710
+ "loss": 2.8587716674804686,
1711
+ "step": 21900
1712
+ },
1713
+ {
1714
+ "epoch": 0.2,
1715
+ "grad_norm": 0.1462126523256302,
1716
+ "learning_rate": 0.0009055230800285523,
1717
+ "loss": 2.854002685546875,
1718
+ "step": 22000
1719
+ },
1720
+ {
1721
+ "epoch": 0.2,
1722
+ "eval_loss": 3.1955888271331787,
1723
+ "eval_runtime": 6.8048,
1724
+ "eval_samples_per_second": 84.646,
1725
+ "eval_steps_per_second": 21.161,
1726
+ "step": 22000
1727
+ },
1728
+ {
1729
+ "epoch": 0.2009090909090909,
1730
+ "grad_norm": 0.1481468826532364,
1731
+ "learning_rate": 0.0009046849295434343,
1732
+ "loss": 2.848748779296875,
1733
+ "step": 22100
1734
+ },
1735
+ {
1736
+ "epoch": 0.2018181818181818,
1737
+ "grad_norm": 0.13478662073612213,
1738
+ "learning_rate": 0.0009038434691469946,
1739
+ "loss": 2.859345703125,
1740
+ "step": 22200
1741
+ },
1742
+ {
1743
+ "epoch": 0.20272727272727273,
1744
+ "grad_norm": 0.1626402586698532,
1745
+ "learning_rate": 0.0009029987057215234,
1746
+ "loss": 2.879791259765625,
1747
+ "step": 22300
1748
+ },
1749
+ {
1750
+ "epoch": 0.20363636363636364,
1751
+ "grad_norm": 0.1763114631175995,
1752
+ "learning_rate": 0.0009021506461763272,
1753
+ "loss": 2.8198529052734376,
1754
+ "step": 22400
1755
+ },
1756
+ {
1757
+ "epoch": 0.20454545454545456,
1758
+ "grad_norm": 0.1592358946800232,
1759
+ "learning_rate": 0.0009012992974476706,
1760
+ "loss": 2.8292422485351563,
1761
+ "step": 22500
1762
+ },
1763
+ {
1764
+ "epoch": 0.20545454545454545,
1765
+ "grad_norm": 0.16374371945858002,
1766
+ "learning_rate": 0.0009004446664987209,
1767
+ "loss": 2.839366149902344,
1768
+ "step": 22600
1769
+ },
1770
+ {
1771
+ "epoch": 0.20636363636363636,
1772
+ "grad_norm": 0.16180269420146942,
1773
+ "learning_rate": 0.0008995867603194905,
1774
+ "loss": 2.824099426269531,
1775
+ "step": 22700
1776
+ },
1777
+ {
1778
+ "epoch": 0.20727272727272728,
1779
+ "grad_norm": 0.1817290335893631,
1780
+ "learning_rate": 0.0008987255859267796,
1781
+ "loss": 2.852536926269531,
1782
+ "step": 22800
1783
+ },
1784
+ {
1785
+ "epoch": 0.2081818181818182,
1786
+ "grad_norm": 0.1729678362607956,
1787
+ "learning_rate": 0.0008978611503641194,
1788
+ "loss": 2.87031005859375,
1789
+ "step": 22900
1790
+ },
1791
+ {
1792
+ "epoch": 0.20909090909090908,
1793
+ "grad_norm": 0.13794034719467163,
1794
+ "learning_rate": 0.000896993460701714,
1795
+ "loss": 2.820664367675781,
1796
+ "step": 23000
1797
+ },
1798
+ {
1799
+ "epoch": 0.20909090909090908,
1800
+ "eval_loss": 3.189971923828125,
1801
+ "eval_runtime": 6.8545,
1802
+ "eval_samples_per_second": 84.032,
1803
+ "eval_steps_per_second": 21.008,
1804
+ "step": 23000
1805
+ },
1806
+ {
1807
+ "epoch": 0.21,
1808
+ "grad_norm": 0.14675737917423248,
1809
+ "learning_rate": 0.0008961225240363826,
1810
+ "loss": 2.834775390625,
1811
+ "step": 23100
1812
+ },
1813
+ {
1814
+ "epoch": 0.2109090909090909,
1815
+ "grad_norm": 0.14198128879070282,
1816
+ "learning_rate": 0.0008952483474915021,
1817
+ "loss": 2.8415087890625,
1818
+ "step": 23200
1819
+ },
1820
+ {
1821
+ "epoch": 0.21181818181818182,
1822
+ "grad_norm": 0.15185648202896118,
1823
+ "learning_rate": 0.0008943709382169476,
1824
+ "loss": 2.8519711303710937,
1825
+ "step": 23300
1826
+ },
1827
+ {
1828
+ "epoch": 0.21272727272727274,
1829
+ "grad_norm": 0.15163777768611908,
1830
+ "learning_rate": 0.0008934903033890352,
1831
+ "loss": 2.8500375366210937,
1832
+ "step": 23400
1833
+ },
1834
+ {
1835
+ "epoch": 0.21363636363636362,
1836
+ "grad_norm": 0.15410897135734558,
1837
+ "learning_rate": 0.0008926064502104623,
1838
+ "loss": 2.8353033447265625,
1839
+ "step": 23500
1840
+ },
1841
+ {
1842
+ "epoch": 0.21454545454545454,
1843
+ "grad_norm": 0.22796989977359772,
1844
+ "learning_rate": 0.0008917193859102497,
1845
+ "loss": 2.8228555297851563,
1846
+ "step": 23600
1847
+ },
1848
+ {
1849
+ "epoch": 0.21545454545454545,
1850
+ "grad_norm": 0.1605440229177475,
1851
+ "learning_rate": 0.0008908291177436813,
1852
+ "loss": 2.8567230224609377,
1853
+ "step": 23700
1854
+ },
1855
+ {
1856
+ "epoch": 0.21636363636363637,
1857
+ "grad_norm": 0.16412964463233948,
1858
+ "learning_rate": 0.0008899356529922459,
1859
+ "loss": 2.8504483032226564,
1860
+ "step": 23800
1861
+ },
1862
+ {
1863
+ "epoch": 0.21727272727272728,
1864
+ "grad_norm": 0.13510747253894806,
1865
+ "learning_rate": 0.0008890389989635767,
1866
+ "loss": 2.87345458984375,
1867
+ "step": 23900
1868
+ },
1869
+ {
1870
+ "epoch": 0.21818181818181817,
1871
+ "grad_norm": 0.14522412419319153,
1872
+ "learning_rate": 0.0008881391629913921,
1873
+ "loss": 2.827217712402344,
1874
+ "step": 24000
1875
+ },
1876
+ {
1877
+ "epoch": 0.21818181818181817,
1878
+ "eval_loss": 3.1835873126983643,
1879
+ "eval_runtime": 6.8315,
1880
+ "eval_samples_per_second": 84.315,
1881
+ "eval_steps_per_second": 21.079,
1882
+ "step": 24000
1883
  }
1884
  ],
1885
  "logging_steps": 100,
 
1899
  "attributes": {}
1900
  }
1901
  },
1902
+ "total_flos": 5.97949943906304e+17,
1903
  "train_batch_size": 22,
1904
  "trial_name": null,
1905
  "trial_params": null