himanshunakrani9 commited on
Commit
c58d6af
·
verified ·
1 Parent(s): 79ef13b

Upload training log: sft_output_stage2_checkpoint-500_trainer_state.json

Browse files
training_logs/sft_output_stage2_checkpoint-500_trainer_state.json ADDED
@@ -0,0 +1,494 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.02432098840496878,
6
+ "eval_steps": 500,
7
+ "global_step": 500,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.00048641976809937553,
14
+ "grad_norm": 10.5,
15
+ "learning_rate": 8.754863813229573e-08,
16
+ "loss": 4.2014,
17
+ "mean_token_accuracy": 0.2661036394536495,
18
+ "num_tokens": 83863.0,
19
+ "step": 10
20
+ },
21
+ {
22
+ "epoch": 0.0009728395361987511,
23
+ "grad_norm": 10.5625,
24
+ "learning_rate": 1.8482490272373542e-07,
25
+ "loss": 4.2378,
26
+ "mean_token_accuracy": 0.2676557971164584,
27
+ "num_tokens": 167707.0,
28
+ "step": 20
29
+ },
30
+ {
31
+ "epoch": 0.0014592593042981267,
32
+ "grad_norm": 7.53125,
33
+ "learning_rate": 2.821011673151751e-07,
34
+ "loss": 4.1997,
35
+ "mean_token_accuracy": 0.2689529391005635,
36
+ "num_tokens": 254535.0,
37
+ "step": 30
38
+ },
39
+ {
40
+ "epoch": 0.0019456790723975021,
41
+ "grad_norm": 9.0625,
42
+ "learning_rate": 3.793774319066148e-07,
43
+ "loss": 4.2259,
44
+ "mean_token_accuracy": 0.2677719397470355,
45
+ "num_tokens": 334693.0,
46
+ "step": 40
47
+ },
48
+ {
49
+ "epoch": 0.002432098840496878,
50
+ "grad_norm": 9.1875,
51
+ "learning_rate": 4.766536964980545e-07,
52
+ "loss": 4.2792,
53
+ "mean_token_accuracy": 0.2641894035041332,
54
+ "num_tokens": 413162.0,
55
+ "step": 50
56
+ },
57
+ {
58
+ "epoch": 0.0029185186085962534,
59
+ "grad_norm": 7.8125,
60
+ "learning_rate": 5.739299610894942e-07,
61
+ "loss": 4.2469,
62
+ "mean_token_accuracy": 0.2663512609899044,
63
+ "num_tokens": 494569.0,
64
+ "step": 60
65
+ },
66
+ {
67
+ "epoch": 0.003404938376695629,
68
+ "grad_norm": 9.3125,
69
+ "learning_rate": 6.712062256809339e-07,
70
+ "loss": 4.2553,
71
+ "mean_token_accuracy": 0.26653255317360164,
72
+ "num_tokens": 577934.0,
73
+ "step": 70
74
+ },
75
+ {
76
+ "epoch": 0.0038913581447950043,
77
+ "grad_norm": 11.125,
78
+ "learning_rate": 7.684824902723737e-07,
79
+ "loss": 4.2167,
80
+ "mean_token_accuracy": 0.2706906918436289,
81
+ "num_tokens": 660791.0,
82
+ "step": 80
83
+ },
84
+ {
85
+ "epoch": 0.00437777791289438,
86
+ "grad_norm": 9.375,
87
+ "learning_rate": 8.657587548638133e-07,
88
+ "loss": 4.2568,
89
+ "mean_token_accuracy": 0.26637437790632246,
90
+ "num_tokens": 742730.0,
91
+ "step": 90
92
+ },
93
+ {
94
+ "epoch": 0.004864197680993756,
95
+ "grad_norm": 11.0625,
96
+ "learning_rate": 9.630350194552531e-07,
97
+ "loss": 4.2296,
98
+ "mean_token_accuracy": 0.27034175898879764,
99
+ "num_tokens": 824640.0,
100
+ "step": 100
101
+ },
102
+ {
103
+ "epoch": 0.005350617449093131,
104
+ "grad_norm": 12.0625,
105
+ "learning_rate": 1.0603112840466926e-06,
106
+ "loss": 4.2102,
107
+ "mean_token_accuracy": 0.2690233327448368,
108
+ "num_tokens": 904458.0,
109
+ "step": 110
110
+ },
111
+ {
112
+ "epoch": 0.005837037217192507,
113
+ "grad_norm": 9.5625,
114
+ "learning_rate": 1.1575875486381325e-06,
115
+ "loss": 4.1164,
116
+ "mean_token_accuracy": 0.2748654978349805,
117
+ "num_tokens": 990844.0,
118
+ "step": 120
119
+ },
120
+ {
121
+ "epoch": 0.006323456985291883,
122
+ "grad_norm": 8.75,
123
+ "learning_rate": 1.2548638132295721e-06,
124
+ "loss": 4.2311,
125
+ "mean_token_accuracy": 0.2725300809368491,
126
+ "num_tokens": 1066753.0,
127
+ "step": 130
128
+ },
129
+ {
130
+ "epoch": 0.006809876753391258,
131
+ "grad_norm": 10.5,
132
+ "learning_rate": 1.3521400778210116e-06,
133
+ "loss": 4.1937,
134
+ "mean_token_accuracy": 0.2694834655150771,
135
+ "num_tokens": 1148453.0,
136
+ "step": 140
137
+ },
138
+ {
139
+ "epoch": 0.0072962965214906335,
140
+ "grad_norm": 7.09375,
141
+ "learning_rate": 1.4494163424124515e-06,
142
+ "loss": 4.2132,
143
+ "mean_token_accuracy": 0.27585016619414093,
144
+ "num_tokens": 1229254.0,
145
+ "step": 150
146
+ },
147
+ {
148
+ "epoch": 0.0077827162895900085,
149
+ "grad_norm": 11.1875,
150
+ "learning_rate": 1.5466926070038912e-06,
151
+ "loss": 4.2072,
152
+ "mean_token_accuracy": 0.2732413921505213,
153
+ "num_tokens": 1309965.0,
154
+ "step": 160
155
+ },
156
+ {
157
+ "epoch": 0.008269136057689384,
158
+ "grad_norm": 6.84375,
159
+ "learning_rate": 1.6439688715953309e-06,
160
+ "loss": 4.2313,
161
+ "mean_token_accuracy": 0.27060455102473496,
162
+ "num_tokens": 1394597.0,
163
+ "step": 170
164
+ },
165
+ {
166
+ "epoch": 0.00875555582578876,
167
+ "grad_norm": 8.6875,
168
+ "learning_rate": 1.7412451361867705e-06,
169
+ "loss": 4.2421,
170
+ "mean_token_accuracy": 0.26859637089073657,
171
+ "num_tokens": 1477121.0,
172
+ "step": 180
173
+ },
174
+ {
175
+ "epoch": 0.009241975593888136,
176
+ "grad_norm": 7.1875,
177
+ "learning_rate": 1.8385214007782104e-06,
178
+ "loss": 4.132,
179
+ "mean_token_accuracy": 0.27875553015619514,
180
+ "num_tokens": 1560466.0,
181
+ "step": 190
182
+ },
183
+ {
184
+ "epoch": 0.009728395361987512,
185
+ "grad_norm": 6.40625,
186
+ "learning_rate": 1.93579766536965e-06,
187
+ "loss": 4.1605,
188
+ "mean_token_accuracy": 0.27900548223406074,
189
+ "num_tokens": 1646243.0,
190
+ "step": 200
191
+ },
192
+ {
193
+ "epoch": 0.010214815130086886,
194
+ "grad_norm": 12.625,
195
+ "learning_rate": 2.0330739299610896e-06,
196
+ "loss": 4.1826,
197
+ "mean_token_accuracy": 0.27863924242556093,
198
+ "num_tokens": 1724202.0,
199
+ "step": 210
200
+ },
201
+ {
202
+ "epoch": 0.010701234898186262,
203
+ "grad_norm": 6.59375,
204
+ "learning_rate": 2.1303501945525293e-06,
205
+ "loss": 4.0799,
206
+ "mean_token_accuracy": 0.28141033593565223,
207
+ "num_tokens": 1808010.0,
208
+ "step": 220
209
+ },
210
+ {
211
+ "epoch": 0.011187654666285638,
212
+ "grad_norm": 5.78125,
213
+ "learning_rate": 2.2276264591439694e-06,
214
+ "loss": 4.1202,
215
+ "mean_token_accuracy": 0.27763844318687914,
216
+ "num_tokens": 1890945.0,
217
+ "step": 230
218
+ },
219
+ {
220
+ "epoch": 0.011674074434385014,
221
+ "grad_norm": 6.3125,
222
+ "learning_rate": 2.3249027237354086e-06,
223
+ "loss": 4.1302,
224
+ "mean_token_accuracy": 0.27731590420007707,
225
+ "num_tokens": 1970319.0,
226
+ "step": 240
227
+ },
228
+ {
229
+ "epoch": 0.01216049420248439,
230
+ "grad_norm": 5.90625,
231
+ "learning_rate": 2.4221789883268483e-06,
232
+ "loss": 4.0835,
233
+ "mean_token_accuracy": 0.28078359123319385,
234
+ "num_tokens": 2048795.0,
235
+ "step": 250
236
+ },
237
+ {
238
+ "epoch": 0.012646913970583765,
239
+ "grad_norm": 6.09375,
240
+ "learning_rate": 2.519455252918288e-06,
241
+ "loss": 4.103,
242
+ "mean_token_accuracy": 0.2803995430469513,
243
+ "num_tokens": 2131083.0,
244
+ "step": 260
245
+ },
246
+ {
247
+ "epoch": 0.01313333373868314,
248
+ "grad_norm": 6.1875,
249
+ "learning_rate": 2.616731517509728e-06,
250
+ "loss": 4.0389,
251
+ "mean_token_accuracy": 0.28429711759090426,
252
+ "num_tokens": 2218851.0,
253
+ "step": 270
254
+ },
255
+ {
256
+ "epoch": 0.013619753506782515,
257
+ "grad_norm": 7.4375,
258
+ "learning_rate": 2.7140077821011673e-06,
259
+ "loss": 4.0774,
260
+ "mean_token_accuracy": 0.27948059868067504,
261
+ "num_tokens": 2298746.0,
262
+ "step": 280
263
+ },
264
+ {
265
+ "epoch": 0.014106173274881891,
266
+ "grad_norm": 6.34375,
267
+ "learning_rate": 2.8112840466926074e-06,
268
+ "loss": 4.0365,
269
+ "mean_token_accuracy": 0.2831402227282524,
270
+ "num_tokens": 2383485.0,
271
+ "step": 290
272
+ },
273
+ {
274
+ "epoch": 0.014592593042981267,
275
+ "grad_norm": 6.28125,
276
+ "learning_rate": 2.908560311284047e-06,
277
+ "loss": 4.0335,
278
+ "mean_token_accuracy": 0.2935697190463543,
279
+ "num_tokens": 2465555.0,
280
+ "step": 300
281
+ },
282
+ {
283
+ "epoch": 0.015079012811080643,
284
+ "grad_norm": 5.96875,
285
+ "learning_rate": 3.0058365758754864e-06,
286
+ "loss": 4.0754,
287
+ "mean_token_accuracy": 0.2854702062904835,
288
+ "num_tokens": 2549673.0,
289
+ "step": 310
290
+ },
291
+ {
292
+ "epoch": 0.015565432579180017,
293
+ "grad_norm": 8.6875,
294
+ "learning_rate": 3.1031128404669265e-06,
295
+ "loss": 4.0758,
296
+ "mean_token_accuracy": 0.284761586971581,
297
+ "num_tokens": 2629418.0,
298
+ "step": 320
299
+ },
300
+ {
301
+ "epoch": 0.016051852347279395,
302
+ "grad_norm": 5.4375,
303
+ "learning_rate": 3.2003891050583657e-06,
304
+ "loss": 3.9983,
305
+ "mean_token_accuracy": 0.28794277347624303,
306
+ "num_tokens": 2716417.0,
307
+ "step": 330
308
+ },
309
+ {
310
+ "epoch": 0.01653827211537877,
311
+ "grad_norm": 6.15625,
312
+ "learning_rate": 3.297665369649806e-06,
313
+ "loss": 4.0446,
314
+ "mean_token_accuracy": 0.28235615976154804,
315
+ "num_tokens": 2800031.0,
316
+ "step": 340
317
+ },
318
+ {
319
+ "epoch": 0.017024691883478143,
320
+ "grad_norm": 4.75,
321
+ "learning_rate": 3.3949416342412455e-06,
322
+ "loss": 4.0271,
323
+ "mean_token_accuracy": 0.28396289646625517,
324
+ "num_tokens": 2882567.0,
325
+ "step": 350
326
+ },
327
+ {
328
+ "epoch": 0.01751111165157752,
329
+ "grad_norm": 5.9375,
330
+ "learning_rate": 3.4922178988326848e-06,
331
+ "loss": 4.0062,
332
+ "mean_token_accuracy": 0.2845515010878444,
333
+ "num_tokens": 2962962.0,
334
+ "step": 360
335
+ },
336
+ {
337
+ "epoch": 0.017997531419676895,
338
+ "grad_norm": 5.6875,
339
+ "learning_rate": 3.589494163424125e-06,
340
+ "loss": 3.9938,
341
+ "mean_token_accuracy": 0.28803673218935727,
342
+ "num_tokens": 3047989.0,
343
+ "step": 370
344
+ },
345
+ {
346
+ "epoch": 0.018483951187776272,
347
+ "grad_norm": 6.53125,
348
+ "learning_rate": 3.6867704280155645e-06,
349
+ "loss": 4.0167,
350
+ "mean_token_accuracy": 0.2940619019791484,
351
+ "num_tokens": 3128197.0,
352
+ "step": 380
353
+ },
354
+ {
355
+ "epoch": 0.018970370955875646,
356
+ "grad_norm": 5.0,
357
+ "learning_rate": 3.7840466926070042e-06,
358
+ "loss": 4.0211,
359
+ "mean_token_accuracy": 0.2942374531179667,
360
+ "num_tokens": 3212352.0,
361
+ "step": 390
362
+ },
363
+ {
364
+ "epoch": 0.019456790723975024,
365
+ "grad_norm": 6.71875,
366
+ "learning_rate": 3.881322957198444e-06,
367
+ "loss": 3.9347,
368
+ "mean_token_accuracy": 0.29957247953861954,
369
+ "num_tokens": 3292295.0,
370
+ "step": 400
371
+ },
372
+ {
373
+ "epoch": 0.019943210492074398,
374
+ "grad_norm": 7.5625,
375
+ "learning_rate": 3.9785992217898836e-06,
376
+ "loss": 3.8966,
377
+ "mean_token_accuracy": 0.30005436614155767,
378
+ "num_tokens": 3375150.0,
379
+ "step": 410
380
+ },
381
+ {
382
+ "epoch": 0.020429630260173772,
383
+ "grad_norm": 6.65625,
384
+ "learning_rate": 4.075875486381323e-06,
385
+ "loss": 3.9126,
386
+ "mean_token_accuracy": 0.3051870077848434,
387
+ "num_tokens": 3457968.0,
388
+ "step": 420
389
+ },
390
+ {
391
+ "epoch": 0.02091605002827315,
392
+ "grad_norm": 5.5,
393
+ "learning_rate": 4.173151750972763e-06,
394
+ "loss": 3.8285,
395
+ "mean_token_accuracy": 0.315464405529201,
396
+ "num_tokens": 3542320.0,
397
+ "step": 430
398
+ },
399
+ {
400
+ "epoch": 0.021402469796372524,
401
+ "grad_norm": 6.21875,
402
+ "learning_rate": 4.270428015564202e-06,
403
+ "loss": 3.866,
404
+ "mean_token_accuracy": 0.31904961802065374,
405
+ "num_tokens": 3622310.0,
406
+ "step": 440
407
+ },
408
+ {
409
+ "epoch": 0.0218888895644719,
410
+ "grad_norm": 4.9375,
411
+ "learning_rate": 4.367704280155642e-06,
412
+ "loss": 3.8579,
413
+ "mean_token_accuracy": 0.3179220326244831,
414
+ "num_tokens": 3702636.0,
415
+ "step": 450
416
+ },
417
+ {
418
+ "epoch": 0.022375309332571276,
419
+ "grad_norm": 5.25,
420
+ "learning_rate": 4.464980544747082e-06,
421
+ "loss": 3.8523,
422
+ "mean_token_accuracy": 0.3214877925813198,
423
+ "num_tokens": 3782123.0,
424
+ "step": 460
425
+ },
426
+ {
427
+ "epoch": 0.02286172910067065,
428
+ "grad_norm": 6.75,
429
+ "learning_rate": 4.562256809338522e-06,
430
+ "loss": 3.8398,
431
+ "mean_token_accuracy": 0.31928886417299507,
432
+ "num_tokens": 3867827.0,
433
+ "step": 470
434
+ },
435
+ {
436
+ "epoch": 0.023348148868770027,
437
+ "grad_norm": 4.4375,
438
+ "learning_rate": 4.659533073929962e-06,
439
+ "loss": 3.8517,
440
+ "mean_token_accuracy": 0.3213398680090904,
441
+ "num_tokens": 3951450.0,
442
+ "step": 480
443
+ },
444
+ {
445
+ "epoch": 0.0238345686368694,
446
+ "grad_norm": 15.3125,
447
+ "learning_rate": 4.756809338521401e-06,
448
+ "loss": 3.7753,
449
+ "mean_token_accuracy": 0.33106248676776884,
450
+ "num_tokens": 4031202.0,
451
+ "step": 490
452
+ },
453
+ {
454
+ "epoch": 0.02432098840496878,
455
+ "grad_norm": 6.1875,
456
+ "learning_rate": 4.854085603112841e-06,
457
+ "loss": 3.8494,
458
+ "mean_token_accuracy": 0.32765379287302493,
459
+ "num_tokens": 4112839.0,
460
+ "step": 500
461
+ },
462
+ {
463
+ "epoch": 0.02432098840496878,
464
+ "eval_loss": 3.804107904434204,
465
+ "eval_mean_token_accuracy": 0.3340388983488083,
466
+ "eval_num_tokens": 4112839.0,
467
+ "eval_runtime": 43.8233,
468
+ "eval_samples_per_second": 151.654,
469
+ "eval_steps_per_second": 37.925,
470
+ "step": 500
471
+ }
472
+ ],
473
+ "logging_steps": 10,
474
+ "max_steps": 41118,
475
+ "num_input_tokens_seen": 0,
476
+ "num_train_epochs": 2,
477
+ "save_steps": 500,
478
+ "stateful_callbacks": {
479
+ "TrainerControl": {
480
+ "args": {
481
+ "should_epoch_stop": false,
482
+ "should_evaluate": false,
483
+ "should_log": false,
484
+ "should_save": true,
485
+ "should_training_stop": false
486
+ },
487
+ "attributes": {}
488
+ }
489
+ },
490
+ "total_flos": 4.075688662120858e+16,
491
+ "train_batch_size": 4,
492
+ "trial_name": null,
493
+ "trial_params": null
494
+ }