CodeIsAbstract commited on
Commit
43fbb47
·
verified ·
1 Parent(s): e218937

Training in progress, step 24000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e851d31aa87e3f44cdb6b12860dfca56948ed93c4c3e52e3a3bc7173fa578a8b
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f097330090d54848c18e4bda429759e1abba010929a3213906365a68cd5e649a
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:df630ecbd687f1359a191cc034b3009e41a878d4ab7fdb04cf787f0bc23ecb0c
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fec1594771eede4505411cf0b3849475f8da63ebefb31deacc4aea691322fb69
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:43b610cc94e504b24cb2bb3bb992068bab81d217de2c8ad61ad01dfb75e98503
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12d604eeadf163b978dd48de387cbebeba84d52ce3a59a4bcf73ab714592e48f
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f23bcbe880608cbeb7ee4f55362f4ac1ed7848d0f02874b239279ca55eb68fab
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4793a1af866198c083bf7fdbba00d5a4fb25389a6bed64d9ca675044e94ad0ae
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.44,
6
  "eval_steps": 1000,
7
- "global_step": 22000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1746,6 +1746,164 @@
1746
  "eval_samples_per_second": 297.832,
1747
  "eval_steps_per_second": 18.72,
1748
  "step": 22000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1749
  }
1750
  ],
1751
  "logging_steps": 100,
@@ -1765,7 +1923,7 @@
1765
  "attributes": {}
1766
  }
1767
  },
1768
- "total_flos": 8.6239139069952e+17,
1769
  "train_batch_size": 120,
1770
  "trial_name": null,
1771
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.48,
6
  "eval_steps": 1000,
7
+ "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1746
  "eval_samples_per_second": 297.832,
1747
  "eval_steps_per_second": 18.72,
1748
  "step": 22000
1749
+ },
1750
+ {
1751
+ "epoch": 0.442,
1752
+ "grad_norm": 0.13086584210395813,
1753
+ "learning_rate": 0.0005932263306841439,
1754
+ "loss": 3.4057171630859373,
1755
+ "step": 22100
1756
+ },
1757
+ {
1758
+ "epoch": 0.444,
1759
+ "grad_norm": 0.14566738903522491,
1760
+ "learning_rate": 0.0005901287109956757,
1761
+ "loss": 3.374288330078125,
1762
+ "step": 22200
1763
+ },
1764
+ {
1765
+ "epoch": 0.446,
1766
+ "grad_norm": 0.13566701114177704,
1767
+ "learning_rate": 0.0005870275117348753,
1768
+ "loss": 3.3892340087890624,
1769
+ "step": 22300
1770
+ },
1771
+ {
1772
+ "epoch": 0.448,
1773
+ "grad_norm": 0.18162570893764496,
1774
+ "learning_rate": 0.0005839228560696758,
1775
+ "loss": 3.417158508300781,
1776
+ "step": 22400
1777
+ },
1778
+ {
1779
+ "epoch": 0.45,
1780
+ "grad_norm": 0.1380554586648941,
1781
+ "learning_rate": 0.0005808148673052863,
1782
+ "loss": 3.385755615234375,
1783
+ "step": 22500
1784
+ },
1785
+ {
1786
+ "epoch": 0.452,
1787
+ "grad_norm": 0.13876518607139587,
1788
+ "learning_rate": 0.0005777036688792934,
1789
+ "loss": 3.3734304809570315,
1790
+ "step": 22600
1791
+ },
1792
+ {
1793
+ "epoch": 0.454,
1794
+ "grad_norm": 0.1451270580291748,
1795
+ "learning_rate": 0.0005745893843567593,
1796
+ "loss": 3.3767324829101564,
1797
+ "step": 22700
1798
+ },
1799
+ {
1800
+ "epoch": 0.456,
1801
+ "grad_norm": 0.14231646060943604,
1802
+ "learning_rate": 0.0005714721374253151,
1803
+ "loss": 3.3700411987304686,
1804
+ "step": 22800
1805
+ },
1806
+ {
1807
+ "epoch": 0.458,
1808
+ "grad_norm": 0.14160802960395813,
1809
+ "learning_rate": 0.0005683520518902468,
1810
+ "loss": 3.3719024658203125,
1811
+ "step": 22900
1812
+ },
1813
+ {
1814
+ "epoch": 0.46,
1815
+ "grad_norm": 0.15433672070503235,
1816
+ "learning_rate": 0.0005652292516695795,
1817
+ "loss": 3.3654217529296875,
1818
+ "step": 23000
1819
+ },
1820
+ {
1821
+ "epoch": 0.46,
1822
+ "eval_accuracy": 0.3727878481747762,
1823
+ "eval_loss": 3.3823931217193604,
1824
+ "eval_runtime": 6.745,
1825
+ "eval_samples_per_second": 287.769,
1826
+ "eval_steps_per_second": 18.087,
1827
+ "step": 23000
1828
+ },
1829
+ {
1830
+ "epoch": 0.462,
1831
+ "grad_norm": 0.13699641823768616,
1832
+ "learning_rate": 0.0005621038607891551,
1833
+ "loss": 3.3590658569335936,
1834
+ "step": 23100
1835
+ },
1836
+ {
1837
+ "epoch": 0.464,
1838
+ "grad_norm": 0.14364606142044067,
1839
+ "learning_rate": 0.0005589760033777068,
1840
+ "loss": 3.3595770263671874,
1841
+ "step": 23200
1842
+ },
1843
+ {
1844
+ "epoch": 0.466,
1845
+ "grad_norm": 0.15491677820682526,
1846
+ "learning_rate": 0.0005558458036619291,
1847
+ "loss": 3.3718606567382814,
1848
+ "step": 23300
1849
+ },
1850
+ {
1851
+ "epoch": 0.468,
1852
+ "grad_norm": 0.14689157903194427,
1853
+ "learning_rate": 0.0005527133859615443,
1854
+ "loss": 3.3682122802734376,
1855
+ "step": 23400
1856
+ },
1857
+ {
1858
+ "epoch": 0.47,
1859
+ "grad_norm": 0.2537364363670349,
1860
+ "learning_rate": 0.0005495788746843641,
1861
+ "loss": 3.3676806640625,
1862
+ "step": 23500
1863
+ },
1864
+ {
1865
+ "epoch": 0.472,
1866
+ "grad_norm": 0.15198677778244019,
1867
+ "learning_rate": 0.0005464423943213493,
1868
+ "loss": 3.3796755981445314,
1869
+ "step": 23600
1870
+ },
1871
+ {
1872
+ "epoch": 0.474,
1873
+ "grad_norm": 0.14707693457603455,
1874
+ "learning_rate": 0.0005433040694416661,
1875
+ "loss": 3.3943267822265626,
1876
+ "step": 23700
1877
+ },
1878
+ {
1879
+ "epoch": 0.476,
1880
+ "grad_norm": 0.13588809967041016,
1881
+ "learning_rate": 0.0005401640246877371,
1882
+ "loss": 3.3692440795898437,
1883
+ "step": 23800
1884
+ },
1885
+ {
1886
+ "epoch": 0.478,
1887
+ "grad_norm": 0.1468556970357895,
1888
+ "learning_rate": 0.0005370223847702919,
1889
+ "loss": 3.35807861328125,
1890
+ "step": 23900
1891
+ },
1892
+ {
1893
+ "epoch": 0.48,
1894
+ "grad_norm": 0.15797235071659088,
1895
+ "learning_rate": 0.0005338792744634145,
1896
+ "loss": 3.3942901611328127,
1897
+ "step": 24000
1898
+ },
1899
+ {
1900
+ "epoch": 0.48,
1901
+ "eval_accuracy": 0.3733514408918275,
1902
+ "eval_loss": 3.375581741333008,
1903
+ "eval_runtime": 7.2715,
1904
+ "eval_samples_per_second": 266.931,
1905
+ "eval_steps_per_second": 16.778,
1906
+ "step": 24000
1907
  }
1908
  ],
1909
  "logging_steps": 100,
 
1923
  "attributes": {}
1924
  }
1925
  },
1926
+ "total_flos": 9.4079060803584e+17,
1927
  "train_batch_size": 120,
1928
  "trial_name": null,
1929
  "trial_params": null