CodeIsAbstract commited on
Commit
c34bf20
·
verified ·
1 Parent(s): 4cbfc65

Training in progress, step 24000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:09db5475ee35a12c8b710afb6e180ab5bfcd8ce48967fc58bd5457a4e3977985
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bfc5c32d531b4d03970087391e1f7ca6763b9d1de362d6cda421b7d124b979c8
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ff60aa4ce7613d8a03c55e84221468eaecbd9f2572db7e051cb5a7fa0b4c3a04
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:baff61bb16847080902923a972f48196a5da070aa7a9295b3b6f30d9c6da65c5
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4281625b7592c11b921e209e5b1131cddcd67ffe5764c2033c0ef0adb2d550bb
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:707f3295d65fc2b08af5a95df451c0c1f230854850d6c8193dd0137a6c028a4e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a0231b5cc777a2809e149f15d91a081c8817fce35708f3c12c17c3012604a372
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:acdbaf996d8220fb05cb901198d39146ebe82d7a748d641503b0d9a4f0d91bc3
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.44,
6
  "eval_steps": 1000,
7
- "global_step": 22000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1746,6 +1746,164 @@
1746
  "eval_samples_per_second": 310.823,
1747
  "eval_steps_per_second": 19.537,
1748
  "step": 22000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1749
  }
1750
  ],
1751
  "logging_steps": 100,
@@ -1765,7 +1923,7 @@
1765
  "attributes": {}
1766
  }
1767
  },
1768
- "total_flos": 6.8981096448e+17,
1769
  "train_batch_size": 120,
1770
  "trial_name": null,
1771
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.48,
6
  "eval_steps": 1000,
7
+ "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1746
  "eval_samples_per_second": 310.823,
1747
  "eval_steps_per_second": 19.537,
1748
  "step": 22000
1749
+ },
1750
+ {
1751
+ "epoch": 0.442,
1752
+ "grad_norm": 0.1982349008321762,
1753
+ "learning_rate": 0.0003559357984104863,
1754
+ "loss": 3.4337942504882815,
1755
+ "step": 22100
1756
+ },
1757
+ {
1758
+ "epoch": 0.444,
1759
+ "grad_norm": 0.19478100538253784,
1760
+ "learning_rate": 0.00035407722659740545,
1761
+ "loss": 3.401671142578125,
1762
+ "step": 22200
1763
+ },
1764
+ {
1765
+ "epoch": 0.446,
1766
+ "grad_norm": 0.20211219787597656,
1767
+ "learning_rate": 0.0003522165070409252,
1768
+ "loss": 3.416873779296875,
1769
+ "step": 22300
1770
+ },
1771
+ {
1772
+ "epoch": 0.448,
1773
+ "grad_norm": 0.23854756355285645,
1774
+ "learning_rate": 0.0003503537136418055,
1775
+ "loss": 3.445356750488281,
1776
+ "step": 22400
1777
+ },
1778
+ {
1779
+ "epoch": 0.45,
1780
+ "grad_norm": 0.1983487904071808,
1781
+ "learning_rate": 0.00034848892038317174,
1782
+ "loss": 3.4137481689453124,
1783
+ "step": 22500
1784
+ },
1785
+ {
1786
+ "epoch": 0.452,
1787
+ "grad_norm": 0.20169700682163239,
1788
+ "learning_rate": 0.00034662220132757596,
1789
+ "loss": 3.4005947875976563,
1790
+ "step": 22600
1791
+ },
1792
+ {
1793
+ "epoch": 0.454,
1794
+ "grad_norm": 0.19934488832950592,
1795
+ "learning_rate": 0.00034475363061405555,
1796
+ "loss": 3.4056222534179685,
1797
+ "step": 22700
1798
+ },
1799
+ {
1800
+ "epoch": 0.456,
1801
+ "grad_norm": 0.212734192609787,
1802
+ "learning_rate": 0.00034288328245518906,
1803
+ "loss": 3.399494323730469,
1804
+ "step": 22800
1805
+ },
1806
+ {
1807
+ "epoch": 0.458,
1808
+ "grad_norm": 0.2278880625963211,
1809
+ "learning_rate": 0.00034101123113414807,
1810
+ "loss": 3.403822326660156,
1811
+ "step": 22900
1812
+ },
1813
+ {
1814
+ "epoch": 0.46,
1815
+ "grad_norm": 0.4838724434375763,
1816
+ "learning_rate": 0.0003391375510017477,
1817
+ "loss": 3.447413024902344,
1818
+ "step": 23000
1819
+ },
1820
+ {
1821
+ "epoch": 0.46,
1822
+ "eval_accuracy": 0.3657575583429366,
1823
+ "eval_loss": 3.438328742980957,
1824
+ "eval_runtime": 6.1806,
1825
+ "eval_samples_per_second": 314.047,
1826
+ "eval_steps_per_second": 19.739,
1827
+ "step": 23000
1828
+ },
1829
+ {
1830
+ "epoch": 0.462,
1831
+ "grad_norm": 0.4579889178276062,
1832
+ "learning_rate": 0.00033726231647349304,
1833
+ "loss": 3.4946719360351564,
1834
+ "step": 23100
1835
+ },
1836
+ {
1837
+ "epoch": 0.464,
1838
+ "grad_norm": 0.724980890750885,
1839
+ "learning_rate": 0.00033538560202662403,
1840
+ "loss": 3.561537170410156,
1841
+ "step": 23200
1842
+ },
1843
+ {
1844
+ "epoch": 0.466,
1845
+ "grad_norm": 1.940724492073059,
1846
+ "learning_rate": 0.00033350748219715745,
1847
+ "loss": 3.64361572265625,
1848
+ "step": 23300
1849
+ },
1850
+ {
1851
+ "epoch": 0.468,
1852
+ "grad_norm": 2.4526569843292236,
1853
+ "learning_rate": 0.00033162803157692656,
1854
+ "loss": 3.734945983886719,
1855
+ "step": 23400
1856
+ },
1857
+ {
1858
+ "epoch": 0.47,
1859
+ "grad_norm": 2.5834765434265137,
1860
+ "learning_rate": 0.0003297473248106184,
1861
+ "loss": 3.86871826171875,
1862
+ "step": 23500
1863
+ },
1864
+ {
1865
+ "epoch": 0.472,
1866
+ "grad_norm": 4.770744323730469,
1867
+ "learning_rate": 0.0003278654365928096,
1868
+ "loss": 3.974892883300781,
1869
+ "step": 23600
1870
+ },
1871
+ {
1872
+ "epoch": 0.474,
1873
+ "grad_norm": 21.92555046081543,
1874
+ "learning_rate": 0.0003259824416649996,
1875
+ "loss": 4.260992431640625,
1876
+ "step": 23700
1877
+ },
1878
+ {
1879
+ "epoch": 0.476,
1880
+ "grad_norm": 3082.5625,
1881
+ "learning_rate": 0.00032409841481264223,
1882
+ "loss": 5.928079833984375,
1883
+ "step": 23800
1884
+ },
1885
+ {
1886
+ "epoch": 0.478,
1887
+ "grad_norm": 3319.344482421875,
1888
+ "learning_rate": 0.0003222134308621751,
1889
+ "loss": 9.030023803710938,
1890
+ "step": 23900
1891
+ },
1892
+ {
1893
+ "epoch": 0.48,
1894
+ "grad_norm": 322.2440490722656,
1895
+ "learning_rate": 0.00032032756467804865,
1896
+ "loss": 8.4674365234375,
1897
+ "step": 24000
1898
+ },
1899
+ {
1900
+ "epoch": 0.48,
1901
+ "eval_accuracy": 0.05309769310108071,
1902
+ "eval_loss": 7.991781711578369,
1903
+ "eval_runtime": 5.9674,
1904
+ "eval_samples_per_second": 325.269,
1905
+ "eval_steps_per_second": 20.445,
1906
+ "step": 24000
1907
  }
1908
  ],
1909
  "logging_steps": 100,
 
1923
  "attributes": {}
1924
  }
1925
  },
1926
+ "total_flos": 7.5252105216e+17,
1927
  "train_batch_size": 120,
1928
  "trial_name": null,
1929
  "trial_params": null