nmthien commited on
Commit
1f109fd
·
verified ·
1 Parent(s): bbb9317

Delete folder last-checkpoint with huggingface_hub

Browse files
last-checkpoint/config.json DELETED
@@ -1,41 +0,0 @@
1
- {
2
- "activation_function": "gelu_new",
3
- "add_cross_attention": false,
4
- "architectures": [
5
- "GPT2LMHeadModel"
6
- ],
7
- "attn_pdrop": 0.1,
8
- "bos_token_id": 0,
9
- "dtype": "float32",
10
- "embd_pdrop": 0.1,
11
- "eos_token_id": 0,
12
- "initializer_range": 0.02,
13
- "layer_norm_epsilon": 1e-05,
14
- "model_type": "gpt2",
15
- "n_ctx": 1024,
16
- "n_embd": 768,
17
- "n_head": 12,
18
- "n_inner": null,
19
- "n_layer": 12,
20
- "n_positions": 1024,
21
- "pad_token_id": 0,
22
- "reorder_and_upcast_attn": false,
23
- "resid_pdrop": 0.1,
24
- "scale_attn_by_inverse_layer_idx": false,
25
- "scale_attn_weights": true,
26
- "summary_activation": null,
27
- "summary_first_dropout": 0.1,
28
- "summary_proj_to_labels": true,
29
- "summary_type": "cls_index",
30
- "summary_use_proj": true,
31
- "task_specific_params": {
32
- "text-generation": {
33
- "do_sample": true,
34
- "max_length": 50
35
- }
36
- },
37
- "tie_word_embeddings": true,
38
- "transformers_version": "5.0.0",
39
- "use_cache": false,
40
- "vocab_size": 32000
41
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
last-checkpoint/generation_config.json DELETED
@@ -1,6 +0,0 @@
1
- {
2
- "_from_model_config": true,
3
- "bos_token_id": 50256,
4
- "eos_token_id": 50256,
5
- "transformers_version": "5.0.0"
6
- }
 
 
 
 
 
 
 
last-checkpoint/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:cfd224ac4f13bb08781434397b33248f9893d360b8636d996957410f8289aae2
3
- size 441688704
 
 
 
 
last-checkpoint/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:23ff7b66d3c3b540fcbef896a735182e8d431fff31c994d58ecafc7bdeb449f4
3
- size 883473803
 
 
 
 
last-checkpoint/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:9449910befbd8e82c9f7eebd9e032fd8d60e664bc2408c9df91eccb93185118e
3
- size 14645
 
 
 
 
last-checkpoint/scaler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:b62db0ba9861d9ab63380744e79a287faa461a1bf55700140a411fe1e976f1cd
3
- size 1383
 
 
 
 
last-checkpoint/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:6b115c78407cd8a402fe92c02d25a01a03ac04d9159fdfdaf5de4d8b081df935
3
- size 1465
 
 
 
 
last-checkpoint/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
last-checkpoint/tokenizer_config.json DELETED
@@ -1,12 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "backend": "tokenizers",
4
- "bos_token": "<|endoftext|>",
5
- "eos_token": "<|endoftext|>",
6
- "errors": "replace",
7
- "is_local": false,
8
- "model_max_length": 1024,
9
- "pad_token": "<|endoftext|>",
10
- "tokenizer_class": "GPT2Tokenizer",
11
- "unk_token": "<|endoftext|>"
12
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
last-checkpoint/trainer_state.json DELETED
@@ -1,736 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.5,
6
- "eval_steps": 500,
7
- "global_step": 4500,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.005555555555555556,
14
- "grad_norm": 0.35775071382522583,
15
- "learning_rate": 0.0001989111111111111,
16
- "loss": 3.360899658203125,
17
- "step": 50
18
- },
19
- {
20
- "epoch": 0.011111111111111112,
21
- "grad_norm": 0.3325912356376648,
22
- "learning_rate": 0.0001978,
23
- "loss": 3.377425537109375,
24
- "step": 100
25
- },
26
- {
27
- "epoch": 0.016666666666666666,
28
- "grad_norm": 0.35807931423187256,
29
- "learning_rate": 0.0001966888888888889,
30
- "loss": 3.375316467285156,
31
- "step": 150
32
- },
33
- {
34
- "epoch": 0.022222222222222223,
35
- "grad_norm": 0.3500971794128418,
36
- "learning_rate": 0.00019557777777777779,
37
- "loss": 3.3736077880859376,
38
- "step": 200
39
- },
40
- {
41
- "epoch": 0.027777777777777776,
42
- "grad_norm": 0.3371691107749939,
43
- "learning_rate": 0.00019446666666666669,
44
- "loss": 3.371192626953125,
45
- "step": 250
46
- },
47
- {
48
- "epoch": 0.03333333333333333,
49
- "grad_norm": 0.3565760552883148,
50
- "learning_rate": 0.00019335555555555556,
51
- "loss": 3.3762380981445315,
52
- "step": 300
53
- },
54
- {
55
- "epoch": 0.03888888888888889,
56
- "grad_norm": 0.338236540555954,
57
- "learning_rate": 0.00019224444444444446,
58
- "loss": 3.2787982177734376,
59
- "step": 350
60
- },
61
- {
62
- "epoch": 0.044444444444444446,
63
- "grad_norm": 0.3267105221748352,
64
- "learning_rate": 0.00019113333333333334,
65
- "loss": 3.1427099609375,
66
- "step": 400
67
- },
68
- {
69
- "epoch": 0.05,
70
- "grad_norm": 0.351743221282959,
71
- "learning_rate": 0.00019002222222222224,
72
- "loss": 3.177709655761719,
73
- "step": 450
74
- },
75
- {
76
- "epoch": 0.05555555555555555,
77
- "grad_norm": 0.3772064745426178,
78
- "learning_rate": 0.00018891111111111114,
79
- "loss": 3.3302297973632813,
80
- "step": 500
81
- },
82
- {
83
- "epoch": 0.05555555555555555,
84
- "eval_loss": 3.2087979316711426,
85
- "eval_runtime": 166.477,
86
- "eval_samples_per_second": 52.187,
87
- "eval_steps_per_second": 1.634,
88
- "step": 500
89
- },
90
- {
91
- "epoch": 0.06111111111111111,
92
- "grad_norm": 0.3273681402206421,
93
- "learning_rate": 0.0001878,
94
- "loss": 3.3618435668945312,
95
- "step": 550
96
- },
97
- {
98
- "epoch": 0.06666666666666667,
99
- "grad_norm": 0.32075268030166626,
100
- "learning_rate": 0.00018668888888888889,
101
- "loss": 3.3656939697265624,
102
- "step": 600
103
- },
104
- {
105
- "epoch": 0.07222222222222222,
106
- "grad_norm": 0.3335159122943878,
107
- "learning_rate": 0.00018557777777777779,
108
- "loss": 3.3647976684570313,
109
- "step": 650
110
- },
111
- {
112
- "epoch": 0.07777777777777778,
113
- "grad_norm": 0.3546631336212158,
114
- "learning_rate": 0.0001844666666666667,
115
- "loss": 3.3779290771484374,
116
- "step": 700
117
- },
118
- {
119
- "epoch": 0.08333333333333333,
120
- "grad_norm": 0.3615032732486725,
121
- "learning_rate": 0.00018335555555555556,
122
- "loss": 3.373221435546875,
123
- "step": 750
124
- },
125
- {
126
- "epoch": 0.08888888888888889,
127
- "grad_norm": 0.3439997434616089,
128
- "learning_rate": 0.00018224444444444446,
129
- "loss": 3.3516156005859377,
130
- "step": 800
131
- },
132
- {
133
- "epoch": 0.09444444444444444,
134
- "grad_norm": 0.3357887268066406,
135
- "learning_rate": 0.00018113333333333334,
136
- "loss": 3.3828070068359377,
137
- "step": 850
138
- },
139
- {
140
- "epoch": 0.1,
141
- "grad_norm": 0.3612283170223236,
142
- "learning_rate": 0.00018002222222222224,
143
- "loss": 3.3401724243164064,
144
- "step": 900
145
- },
146
- {
147
- "epoch": 0.10555555555555556,
148
- "grad_norm": 0.32794448733329773,
149
- "learning_rate": 0.0001789111111111111,
150
- "loss": 3.261454772949219,
151
- "step": 950
152
- },
153
- {
154
- "epoch": 0.1111111111111111,
155
- "grad_norm": 0.3254797160625458,
156
- "learning_rate": 0.0001778,
157
- "loss": 3.1203973388671873,
158
- "step": 1000
159
- },
160
- {
161
- "epoch": 0.1111111111111111,
162
- "eval_loss": 3.1987435817718506,
163
- "eval_runtime": 165.973,
164
- "eval_samples_per_second": 52.346,
165
- "eval_steps_per_second": 1.639,
166
- "step": 1000
167
- },
168
- {
169
- "epoch": 0.11666666666666667,
170
- "grad_norm": 0.3622206151485443,
171
- "learning_rate": 0.0001766888888888889,
172
- "loss": 3.161617431640625,
173
- "step": 1050
174
- },
175
- {
176
- "epoch": 0.12222222222222222,
177
- "grad_norm": 0.381729394197464,
178
- "learning_rate": 0.0001755777777777778,
179
- "loss": 3.3174508666992186,
180
- "step": 1100
181
- },
182
- {
183
- "epoch": 0.12777777777777777,
184
- "grad_norm": 0.3418969213962555,
185
- "learning_rate": 0.00017446666666666666,
186
- "loss": 3.3552154541015624,
187
- "step": 1150
188
- },
189
- {
190
- "epoch": 0.13333333333333333,
191
- "grad_norm": 0.32382291555404663,
192
- "learning_rate": 0.00017335555555555556,
193
- "loss": 3.3517587280273435,
194
- "step": 1200
195
- },
196
- {
197
- "epoch": 0.1388888888888889,
198
- "grad_norm": 0.33188289403915405,
199
- "learning_rate": 0.00017224444444444446,
200
- "loss": 3.3422265625,
201
- "step": 1250
202
- },
203
- {
204
- "epoch": 0.14444444444444443,
205
- "grad_norm": 0.3254759907722473,
206
- "learning_rate": 0.00017113333333333334,
207
- "loss": 3.3487127685546874,
208
- "step": 1300
209
- },
210
- {
211
- "epoch": 0.15,
212
- "grad_norm": 0.34883055090904236,
213
- "learning_rate": 0.00017002222222222224,
214
- "loss": 3.3675653076171876,
215
- "step": 1350
216
- },
217
- {
218
- "epoch": 0.15555555555555556,
219
- "grad_norm": 0.32952237129211426,
220
- "learning_rate": 0.0001689111111111111,
221
- "loss": 3.353576354980469,
222
- "step": 1400
223
- },
224
- {
225
- "epoch": 0.16111111111111112,
226
- "grad_norm": 0.3523924648761749,
227
- "learning_rate": 0.0001678,
228
- "loss": 3.3583038330078123,
229
- "step": 1450
230
- },
231
- {
232
- "epoch": 0.16666666666666666,
233
- "grad_norm": 0.33520838618278503,
234
- "learning_rate": 0.0001666888888888889,
235
- "loss": 3.337117919921875,
236
- "step": 1500
237
- },
238
- {
239
- "epoch": 0.16666666666666666,
240
- "eval_loss": 3.183213710784912,
241
- "eval_runtime": 166.0918,
242
- "eval_samples_per_second": 52.308,
243
- "eval_steps_per_second": 1.638,
244
- "step": 1500
245
- },
246
- {
247
- "epoch": 0.17222222222222222,
248
- "grad_norm": 0.33549970388412476,
249
- "learning_rate": 0.0001655777777777778,
250
- "loss": 3.272804870605469,
251
- "step": 1550
252
- },
253
- {
254
- "epoch": 0.17777777777777778,
255
- "grad_norm": 0.3457739055156708,
256
- "learning_rate": 0.0001644666666666667,
257
- "loss": 3.1095587158203126,
258
- "step": 1600
259
- },
260
- {
261
- "epoch": 0.18333333333333332,
262
- "grad_norm": 0.3696184754371643,
263
- "learning_rate": 0.00016335555555555556,
264
- "loss": 3.150552978515625,
265
- "step": 1650
266
- },
267
- {
268
- "epoch": 0.18888888888888888,
269
- "grad_norm": 0.33659297227859497,
270
- "learning_rate": 0.00016224444444444444,
271
- "loss": 3.3054623413085937,
272
- "step": 1700
273
- },
274
- {
275
- "epoch": 0.19444444444444445,
276
- "grad_norm": 0.33346593379974365,
277
- "learning_rate": 0.00016113333333333334,
278
- "loss": 3.3228750610351563,
279
- "step": 1750
280
- },
281
- {
282
- "epoch": 0.2,
283
- "grad_norm": 0.3283025622367859,
284
- "learning_rate": 0.00016002222222222224,
285
- "loss": 3.3500897216796877,
286
- "step": 1800
287
- },
288
- {
289
- "epoch": 0.20555555555555555,
290
- "grad_norm": 0.33104243874549866,
291
- "learning_rate": 0.0001589111111111111,
292
- "loss": 3.3303262329101564,
293
- "step": 1850
294
- },
295
- {
296
- "epoch": 0.2111111111111111,
297
- "grad_norm": 0.341381311416626,
298
- "learning_rate": 0.00015780000000000001,
299
- "loss": 3.334217224121094,
300
- "step": 1900
301
- },
302
- {
303
- "epoch": 0.21666666666666667,
304
- "grad_norm": 0.35246169567108154,
305
- "learning_rate": 0.00015668888888888891,
306
- "loss": 3.3466201782226563,
307
- "step": 1950
308
- },
309
- {
310
- "epoch": 0.2222222222222222,
311
- "grad_norm": 0.35079947113990784,
312
- "learning_rate": 0.0001555777777777778,
313
- "loss": 3.3411221313476562,
314
- "step": 2000
315
- },
316
- {
317
- "epoch": 0.2222222222222222,
318
- "eval_loss": 3.173346757888794,
319
- "eval_runtime": 166.0597,
320
- "eval_samples_per_second": 52.319,
321
- "eval_steps_per_second": 1.638,
322
- "step": 2000
323
- },
324
- {
325
- "epoch": 0.22777777777777777,
326
- "grad_norm": 0.3230424225330353,
327
- "learning_rate": 0.00015446666666666666,
328
- "loss": 3.32146484375,
329
- "step": 2050
330
- },
331
- {
332
- "epoch": 0.23333333333333334,
333
- "grad_norm": 0.31917229294776917,
334
- "learning_rate": 0.00015335555555555556,
335
- "loss": 3.3351910400390623,
336
- "step": 2100
337
- },
338
- {
339
- "epoch": 0.2388888888888889,
340
- "grad_norm": 0.3427633047103882,
341
- "learning_rate": 0.00015224444444444446,
342
- "loss": 3.281336364746094,
343
- "step": 2150
344
- },
345
- {
346
- "epoch": 0.24444444444444444,
347
- "grad_norm": 0.3299531042575836,
348
- "learning_rate": 0.00015113333333333334,
349
- "loss": 3.096239013671875,
350
- "step": 2200
351
- },
352
- {
353
- "epoch": 0.25,
354
- "grad_norm": 0.35307979583740234,
355
- "learning_rate": 0.0001500222222222222,
356
- "loss": 3.131253967285156,
357
- "step": 2250
358
- },
359
- {
360
- "epoch": 0.25555555555555554,
361
- "grad_norm": 0.336974561214447,
362
- "learning_rate": 0.00014891111111111111,
363
- "loss": 3.286785888671875,
364
- "step": 2300
365
- },
366
- {
367
- "epoch": 0.2611111111111111,
368
- "grad_norm": 0.35140281915664673,
369
- "learning_rate": 0.00014780000000000001,
370
- "loss": 3.3154971313476564,
371
- "step": 2350
372
- },
373
- {
374
- "epoch": 0.26666666666666666,
375
- "grad_norm": 0.33435744047164917,
376
- "learning_rate": 0.0001466888888888889,
377
- "loss": 3.3314889526367186,
378
- "step": 2400
379
- },
380
- {
381
- "epoch": 0.2722222222222222,
382
- "grad_norm": 0.32398200035095215,
383
- "learning_rate": 0.0001455777777777778,
384
- "loss": 3.3115130615234376,
385
- "step": 2450
386
- },
387
- {
388
- "epoch": 0.2777777777777778,
389
- "grad_norm": 0.3433283269405365,
390
- "learning_rate": 0.0001444666666666667,
391
- "loss": 3.3148236083984375,
392
- "step": 2500
393
- },
394
- {
395
- "epoch": 0.2777777777777778,
396
- "eval_loss": 3.1662325859069824,
397
- "eval_runtime": 165.919,
398
- "eval_samples_per_second": 52.363,
399
- "eval_steps_per_second": 1.639,
400
- "step": 2500
401
- },
402
- {
403
- "epoch": 0.2833333333333333,
404
- "grad_norm": 0.34796038269996643,
405
- "learning_rate": 0.00014335555555555556,
406
- "loss": 3.319035949707031,
407
- "step": 2550
408
- },
409
- {
410
- "epoch": 0.28888888888888886,
411
- "grad_norm": 0.3492030203342438,
412
- "learning_rate": 0.00014224444444444444,
413
- "loss": 3.3430206298828127,
414
- "step": 2600
415
- },
416
- {
417
- "epoch": 0.29444444444444445,
418
- "grad_norm": 0.3453795909881592,
419
- "learning_rate": 0.00014113333333333334,
420
- "loss": 3.3211288452148438,
421
- "step": 2650
422
- },
423
- {
424
- "epoch": 0.3,
425
- "grad_norm": 0.3202071785926819,
426
- "learning_rate": 0.00014002222222222224,
427
- "loss": 3.31465087890625,
428
- "step": 2700
429
- },
430
- {
431
- "epoch": 0.3055555555555556,
432
- "grad_norm": 0.3388383984565735,
433
- "learning_rate": 0.00013891111111111111,
434
- "loss": 3.3147201538085938,
435
- "step": 2750
436
- },
437
- {
438
- "epoch": 0.3111111111111111,
439
- "grad_norm": 0.3404223322868347,
440
- "learning_rate": 0.0001378,
441
- "loss": 3.306570739746094,
442
- "step": 2800
443
- },
444
- {
445
- "epoch": 0.31666666666666665,
446
- "grad_norm": 0.3333878517150879,
447
- "learning_rate": 0.0001366888888888889,
448
- "loss": 3.3179815673828124,
449
- "step": 2850
450
- },
451
- {
452
- "epoch": 0.32222222222222224,
453
- "grad_norm": 0.34150388836860657,
454
- "learning_rate": 0.0001355777777777778,
455
- "loss": 3.321144714355469,
456
- "step": 2900
457
- },
458
- {
459
- "epoch": 0.3277777777777778,
460
- "grad_norm": 0.3324436545372009,
461
- "learning_rate": 0.00013446666666666666,
462
- "loss": 3.3187985229492187,
463
- "step": 2950
464
- },
465
- {
466
- "epoch": 0.3333333333333333,
467
- "grad_norm": 0.3744795024394989,
468
- "learning_rate": 0.00013335555555555557,
469
- "loss": 3.3278173828125,
470
- "step": 3000
471
- },
472
- {
473
- "epoch": 0.3333333333333333,
474
- "eval_loss": 3.1569581031799316,
475
- "eval_runtime": 166.2587,
476
- "eval_samples_per_second": 52.256,
477
- "eval_steps_per_second": 1.636,
478
- "step": 3000
479
- },
480
- {
481
- "epoch": 0.3388888888888889,
482
- "grad_norm": 0.3463502526283264,
483
- "learning_rate": 0.00013224444444444447,
484
- "loss": 3.174356689453125,
485
- "step": 3050
486
- },
487
- {
488
- "epoch": 0.34444444444444444,
489
- "grad_norm": 0.31703874468803406,
490
- "learning_rate": 0.00013113333333333334,
491
- "loss": 3.082904052734375,
492
- "step": 3100
493
- },
494
- {
495
- "epoch": 0.35,
496
- "grad_norm": 0.35603103041648865,
497
- "learning_rate": 0.00013002222222222221,
498
- "loss": 3.2023150634765627,
499
- "step": 3150
500
- },
501
- {
502
- "epoch": 0.35555555555555557,
503
- "grad_norm": 0.33750680088996887,
504
- "learning_rate": 0.00012891111111111112,
505
- "loss": 3.2860113525390626,
506
- "step": 3200
507
- },
508
- {
509
- "epoch": 0.3611111111111111,
510
- "grad_norm": 0.33665114641189575,
511
- "learning_rate": 0.00012780000000000002,
512
- "loss": 3.2967050170898435,
513
- "step": 3250
514
- },
515
- {
516
- "epoch": 0.36666666666666664,
517
- "grad_norm": 0.3330186605453491,
518
- "learning_rate": 0.0001266888888888889,
519
- "loss": 3.337146911621094,
520
- "step": 3300
521
- },
522
- {
523
- "epoch": 0.37222222222222223,
524
- "grad_norm": 0.3167618215084076,
525
- "learning_rate": 0.0001255777777777778,
526
- "loss": 3.247300109863281,
527
- "step": 3350
528
- },
529
- {
530
- "epoch": 0.37777777777777777,
531
- "grad_norm": 0.33100470900535583,
532
- "learning_rate": 0.00012446666666666667,
533
- "loss": 3.0897637939453126,
534
- "step": 3400
535
- },
536
- {
537
- "epoch": 0.38333333333333336,
538
- "grad_norm": 0.33727842569351196,
539
- "learning_rate": 0.00012335555555555557,
540
- "loss": 3.1006646728515626,
541
- "step": 3450
542
- },
543
- {
544
- "epoch": 0.3888888888888889,
545
- "grad_norm": 0.339106023311615,
546
- "learning_rate": 0.00012224444444444444,
547
- "loss": 3.2851516723632814,
548
- "step": 3500
549
- },
550
- {
551
- "epoch": 0.3888888888888889,
552
- "eval_loss": 3.154163360595703,
553
- "eval_runtime": 167.1636,
554
- "eval_samples_per_second": 51.973,
555
- "eval_steps_per_second": 1.627,
556
- "step": 3500
557
- },
558
- {
559
- "epoch": 0.39444444444444443,
560
- "grad_norm": 0.3349100649356842,
561
- "learning_rate": 0.00012113333333333334,
562
- "loss": 3.2821334838867187,
563
- "step": 3550
564
- },
565
- {
566
- "epoch": 0.4,
567
- "grad_norm": 0.3199748396873474,
568
- "learning_rate": 0.00012002222222222224,
569
- "loss": 3.3088592529296874,
570
- "step": 3600
571
- },
572
- {
573
- "epoch": 0.40555555555555556,
574
- "grad_norm": 0.33304205536842346,
575
- "learning_rate": 0.0001189111111111111,
576
- "loss": 3.312510986328125,
577
- "step": 3650
578
- },
579
- {
580
- "epoch": 0.4111111111111111,
581
- "grad_norm": 0.3239525854587555,
582
- "learning_rate": 0.0001178,
583
- "loss": 3.3061819458007813,
584
- "step": 3700
585
- },
586
- {
587
- "epoch": 0.4166666666666667,
588
- "grad_norm": 0.33456292748451233,
589
- "learning_rate": 0.00011668888888888889,
590
- "loss": 3.3097720336914063,
591
- "step": 3750
592
- },
593
- {
594
- "epoch": 0.4222222222222222,
595
- "grad_norm": 0.366735577583313,
596
- "learning_rate": 0.00011557777777777778,
597
- "loss": 3.304618225097656,
598
- "step": 3800
599
- },
600
- {
601
- "epoch": 0.42777777777777776,
602
- "grad_norm": 0.3458787202835083,
603
- "learning_rate": 0.00011446666666666668,
604
- "loss": 3.305858154296875,
605
- "step": 3850
606
- },
607
- {
608
- "epoch": 0.43333333333333335,
609
- "grad_norm": 0.3511033058166504,
610
- "learning_rate": 0.00011335555555555557,
611
- "loss": 3.3072021484375,
612
- "step": 3900
613
- },
614
- {
615
- "epoch": 0.4388888888888889,
616
- "grad_norm": 0.32906875014305115,
617
- "learning_rate": 0.00011224444444444444,
618
- "loss": 3.261636047363281,
619
- "step": 3950
620
- },
621
- {
622
- "epoch": 0.4444444444444444,
623
- "grad_norm": 0.3258017301559448,
624
- "learning_rate": 0.00011113333333333333,
625
- "loss": 3.077975158691406,
626
- "step": 4000
627
- },
628
- {
629
- "epoch": 0.4444444444444444,
630
- "eval_loss": 3.1473567485809326,
631
- "eval_runtime": 166.5794,
632
- "eval_samples_per_second": 52.155,
633
- "eval_steps_per_second": 1.633,
634
- "step": 4000
635
- },
636
- {
637
- "epoch": 0.45,
638
- "grad_norm": 0.3296092450618744,
639
- "learning_rate": 0.00011002222222222223,
640
- "loss": 3.096522216796875,
641
- "step": 4050
642
- },
643
- {
644
- "epoch": 0.45555555555555555,
645
- "grad_norm": 0.36984893679618835,
646
- "learning_rate": 0.00010891111111111112,
647
- "loss": 3.272289123535156,
648
- "step": 4100
649
- },
650
- {
651
- "epoch": 0.46111111111111114,
652
- "grad_norm": 0.3618357479572296,
653
- "learning_rate": 0.00010780000000000002,
654
- "loss": 3.2838409423828123,
655
- "step": 4150
656
- },
657
- {
658
- "epoch": 0.4666666666666667,
659
- "grad_norm": 0.34125998616218567,
660
- "learning_rate": 0.0001066888888888889,
661
- "loss": 3.3192120361328126,
662
- "step": 4200
663
- },
664
- {
665
- "epoch": 0.4722222222222222,
666
- "grad_norm": 0.3505809009075165,
667
- "learning_rate": 0.00010557777777777778,
668
- "loss": 3.3169589233398438,
669
- "step": 4250
670
- },
671
- {
672
- "epoch": 0.4777777777777778,
673
- "grad_norm": 0.3265681564807892,
674
- "learning_rate": 0.00010446666666666667,
675
- "loss": 3.29230224609375,
676
- "step": 4300
677
- },
678
- {
679
- "epoch": 0.48333333333333334,
680
- "grad_norm": 0.3397888243198395,
681
- "learning_rate": 0.00010335555555555555,
682
- "loss": 3.28514404296875,
683
- "step": 4350
684
- },
685
- {
686
- "epoch": 0.4888888888888889,
687
- "grad_norm": 0.35736849904060364,
688
- "learning_rate": 0.00010224444444444446,
689
- "loss": 3.308734130859375,
690
- "step": 4400
691
- },
692
- {
693
- "epoch": 0.49444444444444446,
694
- "grad_norm": 0.35276156663894653,
695
- "learning_rate": 0.00010113333333333334,
696
- "loss": 3.2899267578125,
697
- "step": 4450
698
- },
699
- {
700
- "epoch": 0.5,
701
- "grad_norm": 0.3468235433101654,
702
- "learning_rate": 0.00010002222222222222,
703
- "loss": 3.3065914916992187,
704
- "step": 4500
705
- },
706
- {
707
- "epoch": 0.5,
708
- "eval_loss": 3.1372437477111816,
709
- "eval_runtime": 165.9576,
710
- "eval_samples_per_second": 52.351,
711
- "eval_steps_per_second": 1.639,
712
- "step": 4500
713
- }
714
- ],
715
- "logging_steps": 50,
716
- "max_steps": 9000,
717
- "num_input_tokens_seen": 0,
718
- "num_train_epochs": 9223372036854775807,
719
- "save_steps": 500,
720
- "stateful_callbacks": {
721
- "TrainerControl": {
722
- "args": {
723
- "should_epoch_stop": false,
724
- "should_evaluate": false,
725
- "should_log": false,
726
- "should_save": true,
727
- "should_training_stop": false
728
- },
729
- "attributes": {}
730
- }
731
- },
732
- "total_flos": 7.5252105216e+16,
733
- "train_batch_size": 32,
734
- "trial_name": null,
735
- "trial_params": null
736
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
last-checkpoint/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:dd37d0c02b2ccb9edd98dd633c884ff4bb66f6f9b90d58062d0802f2813269e2
3
- size 5265