CodeIsAbstract commited on
Commit
6829620
·
verified ·
1 Parent(s): b4cb6cd

Training in progress, step 26000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bfc5c32d531b4d03970087391e1f7ca6763b9d1de362d6cda421b7d124b979c8
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:868ebbe6082f4ffc7e26ac72b39865598fd064bb047080552522aaec1058ad6e
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:baff61bb16847080902923a972f48196a5da070aa7a9295b3b6f30d9c6da65c5
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:958faabaf0650341769a72f6210e71dac68bc179b0667c9cbcc58bde08e00373
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:707f3295d65fc2b08af5a95df451c0c1f230854850d6c8193dd0137a6c028a4e
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df2f80d39b1989ec660dedb19baf83fbbecae216591a339f013067bf7a57f050
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:acdbaf996d8220fb05cb901198d39146ebe82d7a748d641503b0d9a4f0d91bc3
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:56ea06e88c8e616d19485413e7e9adaf9b94b71f0e4fc7adfe79acd6535722fe
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.48,
6
  "eval_steps": 1000,
7
- "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1904,6 +1904,164 @@
1904
  "eval_samples_per_second": 325.269,
1905
  "eval_steps_per_second": 20.445,
1906
  "step": 24000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1907
  }
1908
  ],
1909
  "logging_steps": 100,
@@ -1923,7 +2081,7 @@
1923
  "attributes": {}
1924
  }
1925
  },
1926
- "total_flos": 7.5252105216e+17,
1927
  "train_batch_size": 120,
1928
  "trial_name": null,
1929
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.52,
6
  "eval_steps": 1000,
7
+ "global_step": 26000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1904
  "eval_samples_per_second": 325.269,
1905
  "eval_steps_per_second": 20.445,
1906
  "step": 24000
1907
+ },
1908
+ {
1909
+ "epoch": 0.482,
1910
+ "grad_norm": 338.5876770019531,
1911
+ "learning_rate": 0.0003184408911597519,
1912
+ "loss": 7.839384155273438,
1913
+ "step": 24100
1914
+ },
1915
+ {
1916
+ "epoch": 0.484,
1917
+ "grad_norm": 79.88603210449219,
1918
+ "learning_rate": 0.0003165534852388384,
1919
+ "loss": 7.44333251953125,
1920
+ "step": 24200
1921
+ },
1922
+ {
1923
+ "epoch": 0.486,
1924
+ "grad_norm": 32.77883529663086,
1925
+ "learning_rate": 0.0003146654218759499,
1926
+ "loss": 7.235584106445312,
1927
+ "step": 24300
1928
+ },
1929
+ {
1930
+ "epoch": 0.488,
1931
+ "grad_norm": 27.993099212646484,
1932
+ "learning_rate": 0.0003127767760578392,
1933
+ "loss": 7.177199096679687,
1934
+ "step": 24400
1935
+ },
1936
+ {
1937
+ "epoch": 0.49,
1938
+ "grad_norm": 15.215092658996582,
1939
+ "learning_rate": 0.000310887622794392,
1940
+ "loss": 7.115394287109375,
1941
+ "step": 24500
1942
+ },
1943
+ {
1944
+ "epoch": 0.492,
1945
+ "grad_norm": 12.316529273986816,
1946
+ "learning_rate": 0.00030899803711564806,
1947
+ "loss": 6.990827026367188,
1948
+ "step": 24600
1949
+ },
1950
+ {
1951
+ "epoch": 0.494,
1952
+ "grad_norm": 2.696770191192627,
1953
+ "learning_rate": 0.0003071080940688207,
1954
+ "loss": 7.0589227294921875,
1955
+ "step": 24700
1956
+ },
1957
+ {
1958
+ "epoch": 0.496,
1959
+ "grad_norm": 48.15589141845703,
1960
+ "learning_rate": 0.0003052178687153168,
1961
+ "loss": 7.067171020507812,
1962
+ "step": 24800
1963
+ },
1964
+ {
1965
+ "epoch": 0.498,
1966
+ "grad_norm": 28.01230239868164,
1967
+ "learning_rate": 0.0003033274361277553,
1968
+ "loss": 7.030769653320313,
1969
+ "step": 24900
1970
+ },
1971
+ {
1972
+ "epoch": 0.5,
1973
+ "grad_norm": 69.79560852050781,
1974
+ "learning_rate": 0.0003014368713869857,
1975
+ "loss": 6.974198608398438,
1976
+ "step": 25000
1977
+ },
1978
+ {
1979
+ "epoch": 0.5,
1980
+ "eval_accuracy": 0.09623723724632026,
1981
+ "eval_loss": 6.917368412017822,
1982
+ "eval_runtime": 6.0867,
1983
+ "eval_samples_per_second": 318.891,
1984
+ "eval_steps_per_second": 20.044,
1985
+ "step": 25000
1986
+ },
1987
+ {
1988
+ "epoch": 0.502,
1989
+ "grad_norm": 42.4299201965332,
1990
+ "learning_rate": 0.00029954624957910595,
1991
+ "loss": 6.993025512695312,
1992
+ "step": 25100
1993
+ },
1994
+ {
1995
+ "epoch": 0.504,
1996
+ "grad_norm": 20.528335571289062,
1997
+ "learning_rate": 0.0002976556457924811,
1998
+ "loss": 7.020432739257813,
1999
+ "step": 25200
2000
+ },
2001
+ {
2002
+ "epoch": 0.506,
2003
+ "grad_norm": 42.26169204711914,
2004
+ "learning_rate": 0.0002957651351147596,
2005
+ "loss": 6.934894409179687,
2006
+ "step": 25300
2007
+ },
2008
+ {
2009
+ "epoch": 0.508,
2010
+ "grad_norm": 3.344475507736206,
2011
+ "learning_rate": 0.0002938747926298927,
2012
+ "loss": 6.941484375,
2013
+ "step": 25400
2014
+ },
2015
+ {
2016
+ "epoch": 0.51,
2017
+ "grad_norm": 64.28266906738281,
2018
+ "learning_rate": 0.0002919846934151513,
2019
+ "loss": 7.314700317382813,
2020
+ "step": 25500
2021
+ },
2022
+ {
2023
+ "epoch": 0.512,
2024
+ "grad_norm": 29.82670783996582,
2025
+ "learning_rate": 0.00029009491253814457,
2026
+ "loss": 7.119804077148437,
2027
+ "step": 25600
2028
+ },
2029
+ {
2030
+ "epoch": 0.514,
2031
+ "grad_norm": 59.44496536254883,
2032
+ "learning_rate": 0.0002882055250538384,
2033
+ "loss": 7.006104736328125,
2034
+ "step": 25700
2035
+ },
2036
+ {
2037
+ "epoch": 0.516,
2038
+ "grad_norm": 42.77421569824219,
2039
+ "learning_rate": 0.00028631660600157493,
2040
+ "loss": 6.991202392578125,
2041
+ "step": 25800
2042
+ },
2043
+ {
2044
+ "epoch": 0.518,
2045
+ "grad_norm": 14.600953102111816,
2046
+ "learning_rate": 0.0002844282304020917,
2047
+ "loss": 6.969422607421875,
2048
+ "step": 25900
2049
+ },
2050
+ {
2051
+ "epoch": 0.52,
2052
+ "grad_norm": 77.17425537109375,
2053
+ "learning_rate": 0.00028254047325454234,
2054
+ "loss": 7.046774291992188,
2055
+ "step": 26000
2056
+ },
2057
+ {
2058
+ "epoch": 0.52,
2059
+ "eval_accuracy": 0.06708366478432748,
2060
+ "eval_loss": 7.517740249633789,
2061
+ "eval_runtime": 5.7572,
2062
+ "eval_samples_per_second": 337.143,
2063
+ "eval_steps_per_second": 21.191,
2064
+ "step": 26000
2065
  }
2066
  ],
2067
  "logging_steps": 100,
 
2081
  "attributes": {}
2082
  }
2083
  },
2084
+ "total_flos": 8.1523113984e+17,
2085
  "train_batch_size": 120,
2086
  "trial_name": null,
2087
  "trial_params": null