CodeIsAbstract commited on
Commit
e5f4fce
·
verified ·
1 Parent(s): 25b55d9

Training in progress, step 26000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f097330090d54848c18e4bda429759e1abba010929a3213906365a68cd5e649a
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:766f39cf2aa03b06ce0544465f23f42d796935dadd6619740980847fa379f296
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fec1594771eede4505411cf0b3849475f8da63ebefb31deacc4aea691322fb69
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad971091ff30bc232ebf11b75c27a84c1325701e13bc4a6246fcc2d26e336bb4
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12d604eeadf163b978dd48de387cbebeba84d52ce3a59a4bcf73ab714592e48f
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:232768e06875e03d77948d18a567c6062bc64ee278f8d41937981a714e6a2d53
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4793a1af866198c083bf7fdbba00d5a4fb25389a6bed64d9ca675044e94ad0ae
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2c5343924a6852b4d8168211c95036da6881053e86461e5c21fd663c6b1f4a7b
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.48,
6
  "eval_steps": 1000,
7
- "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1904,6 +1904,164 @@
1904
  "eval_samples_per_second": 266.931,
1905
  "eval_steps_per_second": 16.778,
1906
  "step": 24000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1907
  }
1908
  ],
1909
  "logging_steps": 100,
@@ -1923,7 +2081,7 @@
1923
  "attributes": {}
1924
  }
1925
  },
1926
- "total_flos": 9.4079060803584e+17,
1927
  "train_batch_size": 120,
1928
  "trial_name": null,
1929
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.52,
6
  "eval_steps": 1000,
7
+ "global_step": 26000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1904
  "eval_samples_per_second": 266.931,
1905
  "eval_steps_per_second": 16.778,
1906
  "step": 24000
1907
+ },
1908
+ {
1909
+ "epoch": 0.482,
1910
+ "grad_norm": 0.15001367032527924,
1911
+ "learning_rate": 0.0005307348185995866,
1912
+ "loss": 3.37049560546875,
1913
+ "step": 24100
1914
+ },
1915
+ {
1916
+ "epoch": 0.484,
1917
+ "grad_norm": 0.15163348615169525,
1918
+ "learning_rate": 0.0005275891420647307,
1919
+ "loss": 3.3899249267578124,
1920
+ "step": 24200
1921
+ },
1922
+ {
1923
+ "epoch": 0.486,
1924
+ "grad_norm": 0.13754494488239288,
1925
+ "learning_rate": 0.0005244423697932499,
1926
+ "loss": 3.397990417480469,
1927
+ "step": 24300
1928
+ },
1929
+ {
1930
+ "epoch": 0.488,
1931
+ "grad_norm": 0.13546991348266602,
1932
+ "learning_rate": 0.0005212946267630653,
1933
+ "loss": 3.4154788208007814,
1934
+ "step": 24400
1935
+ },
1936
+ {
1937
+ "epoch": 0.49,
1938
+ "grad_norm": 0.17627890408039093,
1939
+ "learning_rate": 0.0005181460379906534,
1940
+ "loss": 3.4125030517578123,
1941
+ "step": 24500
1942
+ },
1943
+ {
1944
+ "epoch": 0.492,
1945
+ "grad_norm": 0.14234718680381775,
1946
+ "learning_rate": 0.0005149967285260802,
1947
+ "loss": 3.4010299682617187,
1948
+ "step": 24600
1949
+ },
1950
+ {
1951
+ "epoch": 0.494,
1952
+ "grad_norm": 0.1589398831129074,
1953
+ "learning_rate": 0.0005118468234480345,
1954
+ "loss": 3.4132962036132812,
1955
+ "step": 24700
1956
+ },
1957
+ {
1958
+ "epoch": 0.496,
1959
+ "grad_norm": 0.12560983002185822,
1960
+ "learning_rate": 0.0005086964478588614,
1961
+ "loss": 3.3963751220703124,
1962
+ "step": 24800
1963
+ },
1964
+ {
1965
+ "epoch": 0.498,
1966
+ "grad_norm": 0.14177259802818298,
1967
+ "learning_rate": 0.0005055457268795923,
1968
+ "loss": 3.3922305297851563,
1969
+ "step": 24900
1970
+ },
1971
+ {
1972
+ "epoch": 0.5,
1973
+ "grad_norm": 0.13485221564769745,
1974
+ "learning_rate": 0.0005023947856449762,
1975
+ "loss": 3.3956787109375,
1976
+ "step": 25000
1977
+ },
1978
+ {
1979
+ "epoch": 0.5,
1980
+ "eval_accuracy": 0.3757247812423439,
1981
+ "eval_loss": 3.3547720909118652,
1982
+ "eval_runtime": 6.4257,
1983
+ "eval_samples_per_second": 302.069,
1984
+ "eval_steps_per_second": 18.986,
1985
+ "step": 25000
1986
+ },
1987
+ {
1988
+ "epoch": 0.502,
1989
+ "grad_norm": 0.1296602189540863,
1990
+ "learning_rate": 0.00049924374929851,
1991
+ "loss": 3.4078851318359376,
1992
+ "step": 25100
1993
+ },
1994
+ {
1995
+ "epoch": 0.504,
1996
+ "grad_norm": 0.17173421382904053,
1997
+ "learning_rate": 0.0004960927429874685,
1998
+ "loss": 3.378636779785156,
1999
+ "step": 25200
2000
+ },
2001
+ {
2002
+ "epoch": 0.506,
2003
+ "grad_norm": 0.14787626266479492,
2004
+ "learning_rate": 0.0004929418918579327,
2005
+ "loss": 3.3894952392578124,
2006
+ "step": 25300
2007
+ },
2008
+ {
2009
+ "epoch": 0.508,
2010
+ "grad_norm": 0.14193665981292725,
2011
+ "learning_rate": 0.0004897913210498212,
2012
+ "loss": 3.3784576416015626,
2013
+ "step": 25400
2014
+ },
2015
+ {
2016
+ "epoch": 0.51,
2017
+ "grad_norm": 0.34911948442459106,
2018
+ "learning_rate": 0.0004866411556919189,
2019
+ "loss": 3.4079013061523438,
2020
+ "step": 25500
2021
+ },
2022
+ {
2023
+ "epoch": 0.512,
2024
+ "grad_norm": 0.14956624805927277,
2025
+ "learning_rate": 0.00048349152089690765,
2026
+ "loss": 3.395030517578125,
2027
+ "step": 25600
2028
+ },
2029
+ {
2030
+ "epoch": 0.514,
2031
+ "grad_norm": 0.1397118866443634,
2032
+ "learning_rate": 0.0004803425417563974,
2033
+ "loss": 3.3780328369140626,
2034
+ "step": 25700
2035
+ },
2036
+ {
2037
+ "epoch": 0.516,
2038
+ "grad_norm": 0.13487640023231506,
2039
+ "learning_rate": 0.00047719434333595825,
2040
+ "loss": 3.3928378295898436,
2041
+ "step": 25800
2042
+ },
2043
+ {
2044
+ "epoch": 0.518,
2045
+ "grad_norm": 0.14225220680236816,
2046
+ "learning_rate": 0.0004740470506701529,
2047
+ "loss": 3.381436767578125,
2048
+ "step": 25900
2049
+ },
2050
+ {
2051
+ "epoch": 0.52,
2052
+ "grad_norm": 0.13680697977542877,
2053
+ "learning_rate": 0.0004709007887575706,
2054
+ "loss": 3.378739318847656,
2055
+ "step": 26000
2056
+ },
2057
+ {
2058
+ "epoch": 0.52,
2059
+ "eval_accuracy": 0.3769195171452164,
2060
+ "eval_loss": 3.345224142074585,
2061
+ "eval_runtime": 6.5169,
2062
+ "eval_samples_per_second": 297.843,
2063
+ "eval_steps_per_second": 18.721,
2064
+ "step": 26000
2065
  }
2066
  ],
2067
  "logging_steps": 100,
 
2081
  "attributes": {}
2082
  }
2083
  },
2084
+ "total_flos": 1.01918982537216e+18,
2085
  "train_batch_size": 120,
2086
  "trial_name": null,
2087
  "trial_params": null