CodeIsAbstract commited on
Commit
8be48e9
·
verified ·
1 Parent(s): b5a06db

Training in progress, step 25000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7d1c6311d572f46a40ce4eb1f61e2424dd71806261ba704027aaaba0f53be1d3
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:37bd1f1e3132b8f4ca2e1641f76f2ff204353c58ed6c9a8d3a42545f2ef3602e
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:406e804c5931a6000ece6c45c72c2083479252e8a80b808997d89f3582f02d7c
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6260e5ed2a4761dc61809779662239080c0724fec7e1b4b305362821699e054f
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fcb26f09d74efdea654cd6f0c6e78cd3e7623b779d82863159c4f16bc5889ad4
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b75fbbfe47c8eb423183d97d50b297663315c72e52cad7e4622b5579e136e1e
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:929250de26e41d8c2bd1974c52ecfb0856cc75bc17b68237754cbebda202d823
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:149bf5178d2d607b9bd2f9ea479276fe8d71854eead7d7c0885a6ec8bd738dcb
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dd047aae276051c168a6ca7a4dc31647ad0f9fdb6b46bb26cce7430994cbd35b
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a5708c9f5538c42b528454efe7aef875b322a492e6de3be1a1c907f128070c07
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f57c573bc192fc2295ea4932d408c4cde38e64faf388e178f6aa0d97bb4a5608
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73b5a25dd9e3a3d397893fff0e8fc7745468a7cb99cb1e9d5ba28edff7649e26
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e97a8cd9c3363d0ad2e30f9c2a29774c5d7047232697e060cddc7c85b31c0b41
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4880d7f8ee6864023dc6101bec185e3cd189b20c0a83daeb27f6787498653c3b
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3e82abfebcf7dbe10548c9a7640e13daccea206911077357e8a8b41a7d71deaf
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d403fa4d8a99d63710b2370dcfed9351b4870d6631834c55f58ca06f1b5fae15
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7d2ef15daaa8b40793b1d85f9a922559042ab20cd3d4f334c812af308bbdd607
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3b42446608e7d6d7cadc61c46cf9c6f88ed53fed19083dc6f87be5ba5dd29807
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:084371fcc7045b3c000b45241abee083d78c6426437535d482510c54f8bccc10
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac5ab475c9775e1de56aac2a6519852847eef432f44e78e5216dffadd93f4ad6
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9cab327592548328ae2943f7e4c2b7e1f1745457edb6d6e98a239c4ffa106827
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8d5215c1b9e2758e58a91c7d0fa8082001b7513bdb179dca8bdef8dc1f4b134
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.96,
6
  "eval_steps": 1000,
7
- "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -1904,6 +1904,85 @@
1904
  "eval_samples_per_second": 17.652,
1905
  "eval_steps_per_second": 0.565,
1906
  "step": 24000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1907
  }
1908
  ],
1909
  "logging_steps": 100,
@@ -1918,7 +1997,7 @@
1918
  "should_evaluate": false,
1919
  "should_log": false,
1920
  "should_save": true,
1921
- "should_training_stop": false
1922
  },
1923
  "attributes": {}
1924
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
  "eval_steps": 1000,
7
+ "global_step": 25000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
1904
  "eval_samples_per_second": 17.652,
1905
  "eval_steps_per_second": 0.565,
1906
  "step": 24000
1907
+ },
1908
+ {
1909
+ "epoch": 0.964,
1910
+ "grad_norm": 1.4101399183273315,
1911
+ "learning_rate": 1.281599834858893e-05,
1912
+ "loss": 42.2501513671875,
1913
+ "step": 24100
1914
+ },
1915
+ {
1916
+ "epoch": 0.968,
1917
+ "grad_norm": 1.015312910079956,
1918
+ "learning_rate": 1.0131301976646912e-05,
1919
+ "loss": 42.1179931640625,
1920
+ "step": 24200
1921
+ },
1922
+ {
1923
+ "epoch": 0.972,
1924
+ "grad_norm": 0.8839344382286072,
1925
+ "learning_rate": 7.761080465675363e-06,
1926
+ "loss": 42.1094140625,
1927
+ "step": 24300
1928
+ },
1929
+ {
1930
+ "epoch": 0.976,
1931
+ "grad_norm": 1.0040006637573242,
1932
+ "learning_rate": 5.705708400731702e-06,
1933
+ "loss": 42.0864111328125,
1934
+ "step": 24400
1935
+ },
1936
+ {
1937
+ "epoch": 0.98,
1938
+ "grad_norm": 0.9537904262542725,
1939
+ "learning_rate": 3.965510608697098e-06,
1940
+ "loss": 42.1570068359375,
1941
+ "step": 24500
1942
+ },
1943
+ {
1944
+ "epoch": 0.984,
1945
+ "grad_norm": 0.9448596239089966,
1946
+ "learning_rate": 2.5407621069435395e-06,
1947
+ "loss": 42.424609375,
1948
+ "step": 24600
1949
+ },
1950
+ {
1951
+ "epoch": 0.988,
1952
+ "grad_norm": 1.1038907766342163,
1953
+ "learning_rate": 1.4316880598685967e-06,
1954
+ "loss": 42.390634765625,
1955
+ "step": 24700
1956
+ },
1957
+ {
1958
+ "epoch": 0.992,
1959
+ "grad_norm": 0.9914430379867554,
1960
+ "learning_rate": 6.3846374331189e-07,
1961
+ "loss": 42.1272607421875,
1962
+ "step": 24800
1963
+ },
1964
+ {
1965
+ "epoch": 0.996,
1966
+ "grad_norm": 0.954636812210083,
1967
+ "learning_rate": 1.6121451685435774e-07,
1968
+ "loss": 42.1366357421875,
1969
+ "step": 24900
1970
+ },
1971
+ {
1972
+ "epoch": 1.0,
1973
+ "grad_norm": 1.122141718864441,
1974
+ "learning_rate": 1.5804007658104523e-11,
1975
+ "loss": 42.0115185546875,
1976
+ "step": 25000
1977
+ },
1978
+ {
1979
+ "epoch": 1.0,
1980
+ "eval_accuracy": 0.22748093841642228,
1981
+ "eval_loss": 42.03590393066406,
1982
+ "eval_runtime": 6.9933,
1983
+ "eval_samples_per_second": 17.874,
1984
+ "eval_steps_per_second": 0.572,
1985
+ "step": 25000
1986
  }
1987
  ],
1988
  "logging_steps": 100,
 
1997
  "should_evaluate": false,
1998
  "should_log": false,
1999
  "should_save": true,
2000
+ "should_training_stop": true
2001
  },
2002
  "attributes": {}
2003
  }