himanshunakrani9 commited on
Commit
400ef85
·
verified ·
1 Parent(s): f61f588

Upload training log: sft_output_stage1_checkpoint-500_trainer_state.json

Browse files
training_logs/sft_output_stage1_checkpoint-500_trainer_state.json ADDED
@@ -0,0 +1,494 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.310776163468262,
6
+ "eval_steps": 500,
7
+ "global_step": 500,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.00621552326936524,
14
+ "grad_norm": 28.0,
15
+ "learning_rate": 1.1180124223602485e-06,
16
+ "loss": 6.5887,
17
+ "mean_token_accuracy": 0.10676538413390517,
18
+ "num_tokens": 34777.0,
19
+ "step": 10
20
+ },
21
+ {
22
+ "epoch": 0.01243104653873048,
23
+ "grad_norm": 13.875,
24
+ "learning_rate": 2.3602484472049692e-06,
25
+ "loss": 6.4675,
26
+ "mean_token_accuracy": 0.10980516904965043,
27
+ "num_tokens": 72179.0,
28
+ "step": 20
29
+ },
30
+ {
31
+ "epoch": 0.01864656980809572,
32
+ "grad_norm": 17.375,
33
+ "learning_rate": 3.6024844720496897e-06,
34
+ "loss": 6.4813,
35
+ "mean_token_accuracy": 0.11120960218831896,
36
+ "num_tokens": 108227.0,
37
+ "step": 30
38
+ },
39
+ {
40
+ "epoch": 0.02486209307746096,
41
+ "grad_norm": 15.75,
42
+ "learning_rate": 4.84472049689441e-06,
43
+ "loss": 6.4189,
44
+ "mean_token_accuracy": 0.12204592814669013,
45
+ "num_tokens": 143469.0,
46
+ "step": 40
47
+ },
48
+ {
49
+ "epoch": 0.0310776163468262,
50
+ "grad_norm": 13.5,
51
+ "learning_rate": 6.086956521739132e-06,
52
+ "loss": 6.3574,
53
+ "mean_token_accuracy": 0.13387022549286484,
54
+ "num_tokens": 178211.0,
55
+ "step": 50
56
+ },
57
+ {
58
+ "epoch": 0.03729313961619144,
59
+ "grad_norm": 11.625,
60
+ "learning_rate": 7.329192546583852e-06,
61
+ "loss": 6.0724,
62
+ "mean_token_accuracy": 0.16119053736329078,
63
+ "num_tokens": 214904.0,
64
+ "step": 60
65
+ },
66
+ {
67
+ "epoch": 0.04350866288555668,
68
+ "grad_norm": 15.4375,
69
+ "learning_rate": 8.571428571428571e-06,
70
+ "loss": 5.8719,
71
+ "mean_token_accuracy": 0.19856858495622873,
72
+ "num_tokens": 250720.0,
73
+ "step": 70
74
+ },
75
+ {
76
+ "epoch": 0.04972418615492192,
77
+ "grad_norm": 15.125,
78
+ "learning_rate": 9.813664596273292e-06,
79
+ "loss": 5.6144,
80
+ "mean_token_accuracy": 0.22902098931372167,
81
+ "num_tokens": 287003.0,
82
+ "step": 80
83
+ },
84
+ {
85
+ "epoch": 0.05593970942428716,
86
+ "grad_norm": 8.3125,
87
+ "learning_rate": 1.1055900621118014e-05,
88
+ "loss": 5.2115,
89
+ "mean_token_accuracy": 0.2605333041399717,
90
+ "num_tokens": 324727.0,
91
+ "step": 90
92
+ },
93
+ {
94
+ "epoch": 0.0621552326936524,
95
+ "grad_norm": 9.9375,
96
+ "learning_rate": 1.2298136645962735e-05,
97
+ "loss": 4.9736,
98
+ "mean_token_accuracy": 0.281314729526639,
99
+ "num_tokens": 361487.0,
100
+ "step": 100
101
+ },
102
+ {
103
+ "epoch": 0.06837075596301764,
104
+ "grad_norm": 8.3125,
105
+ "learning_rate": 1.3540372670807453e-05,
106
+ "loss": 4.6289,
107
+ "mean_token_accuracy": 0.3168368902057409,
108
+ "num_tokens": 398135.0,
109
+ "step": 110
110
+ },
111
+ {
112
+ "epoch": 0.07458627923238288,
113
+ "grad_norm": 7.59375,
114
+ "learning_rate": 1.4782608695652174e-05,
115
+ "loss": 4.342,
116
+ "mean_token_accuracy": 0.3940485719591379,
117
+ "num_tokens": 433852.0,
118
+ "step": 120
119
+ },
120
+ {
121
+ "epoch": 0.08080180250174812,
122
+ "grad_norm": 9.0625,
123
+ "learning_rate": 1.6024844720496894e-05,
124
+ "loss": 4.0683,
125
+ "mean_token_accuracy": 0.4092902664095163,
126
+ "num_tokens": 469251.0,
127
+ "step": 130
128
+ },
129
+ {
130
+ "epoch": 0.08701732577111336,
131
+ "grad_norm": 6.1875,
132
+ "learning_rate": 1.7267080745341615e-05,
133
+ "loss": 3.9946,
134
+ "mean_token_accuracy": 0.4103577181696892,
135
+ "num_tokens": 505570.0,
136
+ "step": 140
137
+ },
138
+ {
139
+ "epoch": 0.0932328490404786,
140
+ "grad_norm": 7.75,
141
+ "learning_rate": 1.8509316770186337e-05,
142
+ "loss": 3.787,
143
+ "mean_token_accuracy": 0.41990563608706,
144
+ "num_tokens": 541476.0,
145
+ "step": 150
146
+ },
147
+ {
148
+ "epoch": 0.09944837230984384,
149
+ "grad_norm": 8.3125,
150
+ "learning_rate": 1.9751552795031058e-05,
151
+ "loss": 3.5641,
152
+ "mean_token_accuracy": 0.43713102787733077,
153
+ "num_tokens": 576290.0,
154
+ "step": 160
155
+ },
156
+ {
157
+ "epoch": 0.10566389557920908,
158
+ "grad_norm": 6.96875,
159
+ "learning_rate": 1.9999662046926825e-05,
160
+ "loss": 3.5121,
161
+ "mean_token_accuracy": 0.45019695311784746,
162
+ "num_tokens": 612196.0,
163
+ "step": 170
164
+ },
165
+ {
166
+ "epoch": 0.11187941884857432,
167
+ "grad_norm": 9.125,
168
+ "learning_rate": 1.9998289151715887e-05,
169
+ "loss": 3.3835,
170
+ "mean_token_accuracy": 0.4574576035141945,
171
+ "num_tokens": 648857.0,
172
+ "step": 180
173
+ },
174
+ {
175
+ "epoch": 0.11809494211793956,
176
+ "grad_norm": 8.0625,
177
+ "learning_rate": 1.999586033718004e-05,
178
+ "loss": 3.3561,
179
+ "mean_token_accuracy": 0.4481167230755091,
180
+ "num_tokens": 687287.0,
181
+ "step": 190
182
+ },
183
+ {
184
+ "epoch": 0.1243104653873048,
185
+ "grad_norm": 8.5625,
186
+ "learning_rate": 1.999237585982639e-05,
187
+ "loss": 3.2828,
188
+ "mean_token_accuracy": 0.46083246655762194,
189
+ "num_tokens": 723510.0,
190
+ "step": 200
191
+ },
192
+ {
193
+ "epoch": 0.13052598865667003,
194
+ "grad_norm": 7.09375,
195
+ "learning_rate": 1.9987836087650597e-05,
196
+ "loss": 3.2403,
197
+ "mean_token_accuracy": 0.4726243522018194,
198
+ "num_tokens": 759792.0,
199
+ "step": 210
200
+ },
201
+ {
202
+ "epoch": 0.1367415119260353,
203
+ "grad_norm": 7.25,
204
+ "learning_rate": 1.9982241500098e-05,
205
+ "loss": 3.241,
206
+ "mean_token_accuracy": 0.47060870192945004,
207
+ "num_tokens": 797231.0,
208
+ "step": 220
209
+ },
210
+ {
211
+ "epoch": 0.1429570351954005,
212
+ "grad_norm": 5.96875,
213
+ "learning_rate": 1.997559268801299e-05,
214
+ "loss": 3.1573,
215
+ "mean_token_accuracy": 0.4800480019301176,
216
+ "num_tokens": 832234.0,
217
+ "step": 230
218
+ },
219
+ {
220
+ "epoch": 0.14917255846476576,
221
+ "grad_norm": 7.21875,
222
+ "learning_rate": 1.996789035357663e-05,
223
+ "loss": 3.0991,
224
+ "mean_token_accuracy": 0.4860661797225475,
225
+ "num_tokens": 867065.0,
226
+ "step": 240
227
+ },
228
+ {
229
+ "epoch": 0.155388081734131,
230
+ "grad_norm": 6.25,
231
+ "learning_rate": 1.995913531023245e-05,
232
+ "loss": 3.0593,
233
+ "mean_token_accuracy": 0.4916521992534399,
234
+ "num_tokens": 902726.0,
235
+ "step": 250
236
+ },
237
+ {
238
+ "epoch": 0.16160360500349624,
239
+ "grad_norm": 6.25,
240
+ "learning_rate": 1.9949328482600595e-05,
241
+ "loss": 3.1316,
242
+ "mean_token_accuracy": 0.4774193067103624,
243
+ "num_tokens": 940154.0,
244
+ "step": 260
245
+ },
246
+ {
247
+ "epoch": 0.16781912827286147,
248
+ "grad_norm": 7.53125,
249
+ "learning_rate": 1.9938470906380135e-05,
250
+ "loss": 3.0519,
251
+ "mean_token_accuracy": 0.487684416025877,
252
+ "num_tokens": 975119.0,
253
+ "step": 270
254
+ },
255
+ {
256
+ "epoch": 0.17403465154222672,
257
+ "grad_norm": 7.34375,
258
+ "learning_rate": 1.99265637282397e-05,
259
+ "loss": 2.9971,
260
+ "mean_token_accuracy": 0.49169777184724806,
261
+ "num_tokens": 1010842.0,
262
+ "step": 280
263
+ },
264
+ {
265
+ "epoch": 0.18025017481159195,
266
+ "grad_norm": 6.40625,
267
+ "learning_rate": 1.9913608205696385e-05,
268
+ "loss": 3.0361,
269
+ "mean_token_accuracy": 0.4945647891610861,
270
+ "num_tokens": 1046402.0,
271
+ "step": 290
272
+ },
273
+ {
274
+ "epoch": 0.1864656980809572,
275
+ "grad_norm": 6.46875,
276
+ "learning_rate": 1.989960570698294e-05,
277
+ "loss": 3.044,
278
+ "mean_token_accuracy": 0.48813362084329126,
279
+ "num_tokens": 1083030.0,
280
+ "step": 300
281
+ },
282
+ {
283
+ "epoch": 0.19268122135032242,
284
+ "grad_norm": 8.125,
285
+ "learning_rate": 1.988455771090326e-05,
286
+ "loss": 2.9765,
287
+ "mean_token_accuracy": 0.4888417489826679,
288
+ "num_tokens": 1120205.0,
289
+ "step": 310
290
+ },
291
+ {
292
+ "epoch": 0.19889674461968768,
293
+ "grad_norm": 11.4375,
294
+ "learning_rate": 1.986846580667622e-05,
295
+ "loss": 2.9768,
296
+ "mean_token_accuracy": 0.48986743576824665,
297
+ "num_tokens": 1157865.0,
298
+ "step": 320
299
+ },
300
+ {
301
+ "epoch": 0.2051122678890529,
302
+ "grad_norm": 5.75,
303
+ "learning_rate": 1.9851331693767843e-05,
304
+ "loss": 2.9535,
305
+ "mean_token_accuracy": 0.4962737191468477,
306
+ "num_tokens": 1194180.0,
307
+ "step": 330
308
+ },
309
+ {
310
+ "epoch": 0.21132779115841815,
311
+ "grad_norm": 5.71875,
312
+ "learning_rate": 1.9833157181711804e-05,
313
+ "loss": 2.9858,
314
+ "mean_token_accuracy": 0.49163200706243515,
315
+ "num_tokens": 1230432.0,
316
+ "step": 340
317
+ },
318
+ {
319
+ "epoch": 0.21754331442778338,
320
+ "grad_norm": 13.375,
321
+ "learning_rate": 1.9813944189918336e-05,
322
+ "loss": 2.9133,
323
+ "mean_token_accuracy": 0.500376607477665,
324
+ "num_tokens": 1266513.0,
325
+ "step": 350
326
+ },
327
+ {
328
+ "epoch": 0.22375883769714863,
329
+ "grad_norm": 6.0,
330
+ "learning_rate": 1.9793694747471516e-05,
331
+ "loss": 2.877,
332
+ "mean_token_accuracy": 0.5100295204669237,
333
+ "num_tokens": 1301609.0,
334
+ "step": 360
335
+ },
336
+ {
337
+ "epoch": 0.22997436096651386,
338
+ "grad_norm": 6.75,
339
+ "learning_rate": 1.9772410992914973e-05,
340
+ "loss": 2.9227,
341
+ "mean_token_accuracy": 0.49017866142094135,
342
+ "num_tokens": 1339027.0,
343
+ "step": 370
344
+ },
345
+ {
346
+ "epoch": 0.2361898842358791,
347
+ "grad_norm": 11.0,
348
+ "learning_rate": 1.975009517402605e-05,
349
+ "loss": 2.8894,
350
+ "mean_token_accuracy": 0.49963950738310814,
351
+ "num_tokens": 1375035.0,
352
+ "step": 380
353
+ },
354
+ {
355
+ "epoch": 0.24240540750524434,
356
+ "grad_norm": 6.34375,
357
+ "learning_rate": 1.972674964757839e-05,
358
+ "loss": 2.9117,
359
+ "mean_token_accuracy": 0.5014409992843867,
360
+ "num_tokens": 1412736.0,
361
+ "step": 390
362
+ },
363
+ {
364
+ "epoch": 0.2486209307746096,
365
+ "grad_norm": 5.3125,
366
+ "learning_rate": 1.9702376879093057e-05,
367
+ "loss": 2.8421,
368
+ "mean_token_accuracy": 0.504953145608306,
369
+ "num_tokens": 1448784.0,
370
+ "step": 400
371
+ },
372
+ {
373
+ "epoch": 0.2548364540439748,
374
+ "grad_norm": 9.6875,
375
+ "learning_rate": 1.9676979442578158e-05,
376
+ "loss": 2.889,
377
+ "mean_token_accuracy": 0.4983701020479202,
378
+ "num_tokens": 1485680.0,
379
+ "step": 410
380
+ },
381
+ {
382
+ "epoch": 0.26105197731334007,
383
+ "grad_norm": 4.6875,
384
+ "learning_rate": 1.9650560020256974e-05,
385
+ "loss": 2.8382,
386
+ "mean_token_accuracy": 0.5071276389062405,
387
+ "num_tokens": 1522138.0,
388
+ "step": 420
389
+ },
390
+ {
391
+ "epoch": 0.2672675005827053,
392
+ "grad_norm": 6.3125,
393
+ "learning_rate": 1.9623121402284725e-05,
394
+ "loss": 2.9166,
395
+ "mean_token_accuracy": 0.49670529179275036,
396
+ "num_tokens": 1558837.0,
397
+ "step": 430
398
+ },
399
+ {
400
+ "epoch": 0.2734830238520706,
401
+ "grad_norm": 5.46875,
402
+ "learning_rate": 1.9594666486453865e-05,
403
+ "loss": 2.8803,
404
+ "mean_token_accuracy": 0.5076634723693132,
405
+ "num_tokens": 1594354.0,
406
+ "step": 440
407
+ },
408
+ {
409
+ "epoch": 0.27969854712143577,
410
+ "grad_norm": 4.8125,
411
+ "learning_rate": 1.9565198277888086e-05,
412
+ "loss": 2.8316,
413
+ "mean_token_accuracy": 0.5082867383956909,
414
+ "num_tokens": 1629796.0,
415
+ "step": 450
416
+ },
417
+ {
418
+ "epoch": 0.285914070390801,
419
+ "grad_norm": 5.28125,
420
+ "learning_rate": 1.953471988872491e-05,
421
+ "loss": 2.7701,
422
+ "mean_token_accuracy": 0.521901785582304,
423
+ "num_tokens": 1663541.0,
424
+ "step": 460
425
+ },
426
+ {
427
+ "epoch": 0.2921295936601663,
428
+ "grad_norm": 5.03125,
429
+ "learning_rate": 1.950323453778705e-05,
430
+ "loss": 2.801,
431
+ "mean_token_accuracy": 0.5105293065309524,
432
+ "num_tokens": 1699114.0,
433
+ "step": 470
434
+ },
435
+ {
436
+ "epoch": 0.29834511692953153,
437
+ "grad_norm": 5.125,
438
+ "learning_rate": 1.9470745550242428e-05,
439
+ "loss": 2.8536,
440
+ "mean_token_accuracy": 0.5051056195050478,
441
+ "num_tokens": 1736046.0,
442
+ "step": 480
443
+ },
444
+ {
445
+ "epoch": 0.3045606401988967,
446
+ "grad_norm": 6.46875,
447
+ "learning_rate": 1.9437256357253056e-05,
448
+ "loss": 2.8939,
449
+ "mean_token_accuracy": 0.49675586745142936,
450
+ "num_tokens": 1774064.0,
451
+ "step": 490
452
+ },
453
+ {
454
+ "epoch": 0.310776163468262,
455
+ "grad_norm": 6.15625,
456
+ "learning_rate": 1.9402770495612623e-05,
457
+ "loss": 2.7236,
458
+ "mean_token_accuracy": 0.5225549291819334,
459
+ "num_tokens": 1808419.0,
460
+ "step": 500
461
+ },
462
+ {
463
+ "epoch": 0.310776163468262,
464
+ "eval_loss": 2.7760274410247803,
465
+ "eval_mean_token_accuracy": 0.5163302740067927,
466
+ "eval_num_tokens": 1808419.0,
467
+ "eval_runtime": 3.0474,
468
+ "eval_samples_per_second": 170.963,
469
+ "eval_steps_per_second": 42.987,
470
+ "step": 500
471
+ }
472
+ ],
473
+ "logging_steps": 10,
474
+ "max_steps": 3218,
475
+ "num_input_tokens_seen": 0,
476
+ "num_train_epochs": 2,
477
+ "save_steps": 500,
478
+ "stateful_callbacks": {
479
+ "TrainerControl": {
480
+ "args": {
481
+ "should_epoch_stop": false,
482
+ "should_evaluate": false,
483
+ "should_log": false,
484
+ "should_save": true,
485
+ "should_training_stop": false
486
+ },
487
+ "attributes": {}
488
+ }
489
+ },
490
+ "total_flos": 1.778384675561472e+16,
491
+ "train_batch_size": 4,
492
+ "trial_name": null,
493
+ "trial_params": null
494
+ }