CodeIsAbstract commited on
Commit
48bba22
·
verified ·
1 Parent(s): a8a339e

Training in progress, step 68000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4f1052fc44f17677882e9c85aa46ba7cf0ce95d12039334b20adedd2169a62e5
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e65c628bc21a488cc995a999f4ebe043aded7453145c4be01688da0234dc825a
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3fc31a21e830e5240ed3d4e54481bf756f6f044cd5580281c9cff0c9b8736b9e
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a4b726e70e9f1b2d482d4b906ff82dfb782045da636caaeed5e73095f82fd90
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:907db0cce0a5bafbb1926a59c3e73634f18337e15182c3dda71e7ef66bbb3df1
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3cab5a61edbb7c8a2083270aa3aceaa772762aaf6c804e164b525561621e196f
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1845127fa7eca8c5b502f7e5468e1855ba6951691315fb77854957ecb9da8539
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee827426ba5e510531f97228be83448aeb6e47b62607221c379bc9913032527f
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2545454545454545,
6
  "eval_steps": 1000,
7
- "global_step": 64000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -5000,6 +5000,318 @@
5000
  "eval_samples_per_second": 77.085,
5001
  "eval_steps_per_second": 19.271,
5002
  "step": 64000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5003
  }
5004
  ],
5005
  "logging_steps": 100,
@@ -5019,7 +5331,7 @@
5019
  "attributes": {}
5020
  }
5021
  },
5022
- "total_flos": 1.594533183750144e+18,
5023
  "train_batch_size": 22,
5024
  "trial_name": null,
5025
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2909090909090909,
6
  "eval_steps": 1000,
7
+ "global_step": 68000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
5000
  "eval_samples_per_second": 77.085,
5001
  "eval_steps_per_second": 19.271,
5002
  "step": 64000
5003
+ },
5004
+ {
5005
+ "epoch": 0.25545454545454543,
5006
+ "grad_norm": 0.15801353752613068,
5007
+ "learning_rate": 0.0003723892610494738,
5008
+ "loss": 2.6523193359375,
5009
+ "step": 64100
5010
+ },
5011
+ {
5012
+ "epoch": 0.25636363636363635,
5013
+ "grad_norm": 0.16160622239112854,
5014
+ "learning_rate": 0.00037100719429656416,
5015
+ "loss": 2.620086669921875,
5016
+ "step": 64200
5017
+ },
5018
+ {
5019
+ "epoch": 0.25727272727272726,
5020
+ "grad_norm": 0.17042121291160583,
5021
+ "learning_rate": 0.00036962618257367153,
5022
+ "loss": 2.634929504394531,
5023
+ "step": 64300
5024
+ },
5025
+ {
5026
+ "epoch": 0.2581818181818182,
5027
+ "grad_norm": 0.16742800176143646,
5028
+ "learning_rate": 0.00036824623717606755,
5029
+ "loss": 2.6305416870117186,
5030
+ "step": 64400
5031
+ },
5032
+ {
5033
+ "epoch": 0.2590909090909091,
5034
+ "grad_norm": 0.15466608107089996,
5035
+ "learning_rate": 0.000366867369390303,
5036
+ "loss": 2.673544921875,
5037
+ "step": 64500
5038
+ },
5039
+ {
5040
+ "epoch": 0.26,
5041
+ "grad_norm": 0.16298887133598328,
5042
+ "learning_rate": 0.00036548959049411447,
5043
+ "loss": 2.629702453613281,
5044
+ "step": 64600
5045
+ },
5046
+ {
5047
+ "epoch": 0.2609090909090909,
5048
+ "grad_norm": 0.1554904580116272,
5049
+ "learning_rate": 0.00036411291175633275,
5050
+ "loss": 2.664700927734375,
5051
+ "step": 64700
5052
+ },
5053
+ {
5054
+ "epoch": 0.26181818181818184,
5055
+ "grad_norm": 0.16158753633499146,
5056
+ "learning_rate": 0.0003627373444367902,
5057
+ "loss": 2.632271423339844,
5058
+ "step": 64800
5059
+ },
5060
+ {
5061
+ "epoch": 0.26272727272727275,
5062
+ "grad_norm": 0.1622844785451889,
5063
+ "learning_rate": 0.00036136289978622925,
5064
+ "loss": 2.626861877441406,
5065
+ "step": 64900
5066
+ },
5067
+ {
5068
+ "epoch": 0.2636363636363636,
5069
+ "grad_norm": 0.16394828259944916,
5070
+ "learning_rate": 0.0003599895890462101,
5071
+ "loss": 2.648804016113281,
5072
+ "step": 65000
5073
+ },
5074
+ {
5075
+ "epoch": 0.2636363636363636,
5076
+ "eval_loss": 3.0462276935577393,
5077
+ "eval_runtime": 7.4377,
5078
+ "eval_samples_per_second": 77.444,
5079
+ "eval_steps_per_second": 19.361,
5080
+ "step": 65000
5081
+ },
5082
+ {
5083
+ "epoch": 0.26454545454545453,
5084
+ "grad_norm": 0.15556195378303528,
5085
+ "learning_rate": 0.000358617423449018,
5086
+ "loss": 2.6247119140625,
5087
+ "step": 65100
5088
+ },
5089
+ {
5090
+ "epoch": 0.26545454545454544,
5091
+ "grad_norm": 0.1603025197982788,
5092
+ "learning_rate": 0.00035724641421757314,
5093
+ "loss": 2.6453323364257812,
5094
+ "step": 65200
5095
+ },
5096
+ {
5097
+ "epoch": 0.26636363636363636,
5098
+ "grad_norm": 0.17644956707954407,
5099
+ "learning_rate": 0.0003558765725653368,
5100
+ "loss": 2.67004638671875,
5101
+ "step": 65300
5102
+ },
5103
+ {
5104
+ "epoch": 0.2672727272727273,
5105
+ "grad_norm": 0.17642109096050262,
5106
+ "learning_rate": 0.0003545079096962215,
5107
+ "loss": 2.6238552856445314,
5108
+ "step": 65400
5109
+ },
5110
+ {
5111
+ "epoch": 0.2681818181818182,
5112
+ "grad_norm": 0.16550812125205994,
5113
+ "learning_rate": 0.00035314043680449773,
5114
+ "loss": 2.635771484375,
5115
+ "step": 65500
5116
+ },
5117
+ {
5118
+ "epoch": 0.2690909090909091,
5119
+ "grad_norm": 0.18358127772808075,
5120
+ "learning_rate": 0.0003517741650747038,
5121
+ "loss": 2.673054504394531,
5122
+ "step": 65600
5123
+ },
5124
+ {
5125
+ "epoch": 0.27,
5126
+ "grad_norm": 0.14490514993667603,
5127
+ "learning_rate": 0.0003504091056815537,
5128
+ "loss": 2.6274179077148436,
5129
+ "step": 65700
5130
+ },
5131
+ {
5132
+ "epoch": 0.27090909090909093,
5133
+ "grad_norm": 0.15338118374347687,
5134
+ "learning_rate": 0.00034904526978984517,
5135
+ "loss": 2.6343759155273436,
5136
+ "step": 65800
5137
+ },
5138
+ {
5139
+ "epoch": 0.2718181818181818,
5140
+ "grad_norm": 0.1604444533586502,
5141
+ "learning_rate": 0.00034768266855436967,
5142
+ "loss": 2.643642578125,
5143
+ "step": 65900
5144
+ },
5145
+ {
5146
+ "epoch": 0.2727272727272727,
5147
+ "grad_norm": 0.16554522514343262,
5148
+ "learning_rate": 0.0003463213131198198,
5149
+ "loss": 2.645709228515625,
5150
+ "step": 66000
5151
+ },
5152
+ {
5153
+ "epoch": 0.2727272727272727,
5154
+ "eval_loss": 3.0441606044769287,
5155
+ "eval_runtime": 7.4053,
5156
+ "eval_samples_per_second": 77.782,
5157
+ "eval_steps_per_second": 19.445,
5158
+ "step": 66000
5159
+ },
5160
+ {
5161
+ "epoch": 0.2736363636363636,
5162
+ "grad_norm": 0.16071894764900208,
5163
+ "learning_rate": 0.00034496121462069916,
5164
+ "loss": 2.6328436279296876,
5165
+ "step": 66100
5166
+ },
5167
+ {
5168
+ "epoch": 0.27454545454545454,
5169
+ "grad_norm": 0.1610708087682724,
5170
+ "learning_rate": 0.000343602384181231,
5171
+ "loss": 2.650054626464844,
5172
+ "step": 66200
5173
+ },
5174
+ {
5175
+ "epoch": 0.27545454545454545,
5176
+ "grad_norm": 0.1581074744462967,
5177
+ "learning_rate": 0.00034224483291526666,
5178
+ "loss": 2.6330722045898436,
5179
+ "step": 66300
5180
+ },
5181
+ {
5182
+ "epoch": 0.27636363636363637,
5183
+ "grad_norm": 0.16902528703212738,
5184
+ "learning_rate": 0.0003408885719261956,
5185
+ "loss": 2.60602783203125,
5186
+ "step": 66400
5187
+ },
5188
+ {
5189
+ "epoch": 0.2772727272727273,
5190
+ "grad_norm": 0.1647139936685562,
5191
+ "learning_rate": 0.0003395336123068537,
5192
+ "loss": 2.6406063842773437,
5193
+ "step": 66500
5194
+ },
5195
+ {
5196
+ "epoch": 0.2781818181818182,
5197
+ "grad_norm": 0.16128414869308472,
5198
+ "learning_rate": 0.0003381799651394335,
5199
+ "loss": 2.6316217041015624,
5200
+ "step": 66600
5201
+ },
5202
+ {
5203
+ "epoch": 0.2790909090909091,
5204
+ "grad_norm": 0.1957395374774933,
5205
+ "learning_rate": 0.0003368276414953922,
5206
+ "loss": 2.6092684936523436,
5207
+ "step": 66700
5208
+ },
5209
+ {
5210
+ "epoch": 0.28,
5211
+ "grad_norm": 0.14507023990154266,
5212
+ "learning_rate": 0.0003354766524353632,
5213
+ "loss": 2.602089538574219,
5214
+ "step": 66800
5215
+ },
5216
+ {
5217
+ "epoch": 0.2809090909090909,
5218
+ "grad_norm": 0.17211727797985077,
5219
+ "learning_rate": 0.0003341270090090631,
5220
+ "loss": 2.65368896484375,
5221
+ "step": 66900
5222
+ },
5223
+ {
5224
+ "epoch": 0.2818181818181818,
5225
+ "grad_norm": 0.15183623135089874,
5226
+ "learning_rate": 0.0003327787222552032,
5227
+ "loss": 2.647850341796875,
5228
+ "step": 67000
5229
+ },
5230
+ {
5231
+ "epoch": 0.2818181818181818,
5232
+ "eval_loss": 3.037811279296875,
5233
+ "eval_runtime": 7.4649,
5234
+ "eval_samples_per_second": 77.161,
5235
+ "eval_steps_per_second": 19.29,
5236
+ "step": 67000
5237
+ },
5238
+ {
5239
+ "epoch": 0.2827272727272727,
5240
+ "grad_norm": 0.147422194480896,
5241
+ "learning_rate": 0.0003314318032013985,
5242
+ "loss": 2.63864501953125,
5243
+ "step": 67100
5244
+ },
5245
+ {
5246
+ "epoch": 0.28363636363636363,
5247
+ "grad_norm": 0.15864703059196472,
5248
+ "learning_rate": 0.00033008626286407756,
5249
+ "loss": 2.6240463256835938,
5250
+ "step": 67200
5251
+ },
5252
+ {
5253
+ "epoch": 0.28454545454545455,
5254
+ "grad_norm": 0.15877065062522888,
5255
+ "learning_rate": 0.0003287421122483924,
5256
+ "loss": 2.640238952636719,
5257
+ "step": 67300
5258
+ },
5259
+ {
5260
+ "epoch": 0.28545454545454546,
5261
+ "grad_norm": 0.15702606737613678,
5262
+ "learning_rate": 0.00032739936234812863,
5263
+ "loss": 2.6432769775390623,
5264
+ "step": 67400
5265
+ },
5266
+ {
5267
+ "epoch": 0.2863636363636364,
5268
+ "grad_norm": 0.17382535338401794,
5269
+ "learning_rate": 0.0003260580241456155,
5270
+ "loss": 2.6252288818359375,
5271
+ "step": 67500
5272
+ },
5273
+ {
5274
+ "epoch": 0.2872727272727273,
5275
+ "grad_norm": 0.15588997304439545,
5276
+ "learning_rate": 0.00032471810861163573,
5277
+ "loss": 2.6104010009765624,
5278
+ "step": 67600
5279
+ },
5280
+ {
5281
+ "epoch": 0.2881818181818182,
5282
+ "grad_norm": 0.17598041892051697,
5283
+ "learning_rate": 0.0003233796267053365,
5284
+ "loss": 2.6319580078125,
5285
+ "step": 67700
5286
+ },
5287
+ {
5288
+ "epoch": 0.28909090909090907,
5289
+ "grad_norm": 0.15314966440200806,
5290
+ "learning_rate": 0.0003220425893741388,
5291
+ "loss": 2.6283184814453127,
5292
+ "step": 67800
5293
+ },
5294
+ {
5295
+ "epoch": 0.29,
5296
+ "grad_norm": 0.17279887199401855,
5297
+ "learning_rate": 0.000320707007553649,
5298
+ "loss": 2.6345831298828126,
5299
+ "step": 67900
5300
+ },
5301
+ {
5302
+ "epoch": 0.2909090909090909,
5303
+ "grad_norm": 0.20019470155239105,
5304
+ "learning_rate": 0.0003193728921675685,
5305
+ "loss": 2.630293273925781,
5306
+ "step": 68000
5307
+ },
5308
+ {
5309
+ "epoch": 0.2909090909090909,
5310
+ "eval_loss": 3.0448532104492188,
5311
+ "eval_runtime": 7.4367,
5312
+ "eval_samples_per_second": 77.454,
5313
+ "eval_steps_per_second": 19.363,
5314
+ "step": 68000
5315
  }
5316
  ],
5317
  "logging_steps": 100,
 
5331
  "attributes": {}
5332
  }
5333
  },
5334
+ "total_flos": 1.694191507734528e+18,
5335
  "train_batch_size": 22,
5336
  "trial_name": null,
5337
  "trial_params": null