CodeIsAbstract commited on
Commit
dbea085
·
verified ·
1 Parent(s): a312d39

Training in progress, step 300, checkpoint

Browse files
last-checkpoint/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a52e8095da78f763d978476b1110a30523aa854f7c9b9319942cd9d758d4ae7d
3
+ size 1540577008
last-checkpoint/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4adabcf0487be0dab247ad8bee6df0919d714e1b7a53b0b835994eee6673de0e
3
+ size 1695317131
last-checkpoint/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9079206c3c25e14fa2f8a19a24a32c2e5380266de4ccca26cb728acd92fae5e8
3
+ size 14645
last-checkpoint/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c18d3014a31b7d0b10d3631683a1f76c56eca77072280718fe48627ba3a1c6f4
3
+ size 1465
last-checkpoint/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
last-checkpoint/tokenizer_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1000000000,
10
+ "pad_token": "<|padding|>",
11
+ "tokenizer_class": "GPTNeoXTokenizer",
12
+ "trim_offsets": true,
13
+ "unk_token": "<|endoftext|>"
14
+ }
last-checkpoint/trainer_state.json ADDED
@@ -0,0 +1,472 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.06,
6
+ "eval_steps": 150,
7
+ "global_step": 300,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.001,
14
+ "grad_norm": 0.5131137371063232,
15
+ "learning_rate": 6.4e-08,
16
+ "loss": 10.957261657714843,
17
+ "step": 5
18
+ },
19
+ {
20
+ "epoch": 0.002,
21
+ "grad_norm": 0.5448781251907349,
22
+ "learning_rate": 1.44e-07,
23
+ "loss": 10.8884033203125,
24
+ "step": 10
25
+ },
26
+ {
27
+ "epoch": 0.003,
28
+ "grad_norm": 0.5491417646408081,
29
+ "learning_rate": 2.24e-07,
30
+ "loss": 10.823747253417968,
31
+ "step": 15
32
+ },
33
+ {
34
+ "epoch": 0.004,
35
+ "grad_norm": 0.556300163269043,
36
+ "learning_rate": 3.0399999999999997e-07,
37
+ "loss": 10.758411407470703,
38
+ "step": 20
39
+ },
40
+ {
41
+ "epoch": 0.005,
42
+ "grad_norm": 0.6397489309310913,
43
+ "learning_rate": 3.84e-07,
44
+ "loss": 10.67077407836914,
45
+ "step": 25
46
+ },
47
+ {
48
+ "epoch": 0.006,
49
+ "grad_norm": 0.6714406609535217,
50
+ "learning_rate": 4.64e-07,
51
+ "loss": 10.603721618652344,
52
+ "step": 30
53
+ },
54
+ {
55
+ "epoch": 0.007,
56
+ "grad_norm": 0.7375972270965576,
57
+ "learning_rate": 5.44e-07,
58
+ "loss": 10.503799438476562,
59
+ "step": 35
60
+ },
61
+ {
62
+ "epoch": 0.008,
63
+ "grad_norm": 0.6434732675552368,
64
+ "learning_rate": 6.24e-07,
65
+ "loss": 10.438059997558593,
66
+ "step": 40
67
+ },
68
+ {
69
+ "epoch": 0.009,
70
+ "grad_norm": 0.7892774939537048,
71
+ "learning_rate": 7.04e-07,
72
+ "loss": 10.297109222412109,
73
+ "step": 45
74
+ },
75
+ {
76
+ "epoch": 0.01,
77
+ "grad_norm": 0.6715490221977234,
78
+ "learning_rate": 7.84e-07,
79
+ "loss": 10.194837188720703,
80
+ "step": 50
81
+ },
82
+ {
83
+ "epoch": 0.011,
84
+ "grad_norm": 0.644976794719696,
85
+ "learning_rate": 8.639999999999999e-07,
86
+ "loss": 10.094287109375,
87
+ "step": 55
88
+ },
89
+ {
90
+ "epoch": 0.012,
91
+ "grad_norm": 0.643751859664917,
92
+ "learning_rate": 9.439999999999999e-07,
93
+ "loss": 9.97855453491211,
94
+ "step": 60
95
+ },
96
+ {
97
+ "epoch": 0.013,
98
+ "grad_norm": 0.640312910079956,
99
+ "learning_rate": 1.024e-06,
100
+ "loss": 9.856808471679688,
101
+ "step": 65
102
+ },
103
+ {
104
+ "epoch": 0.014,
105
+ "grad_norm": 0.6511930823326111,
106
+ "learning_rate": 1.1040000000000001e-06,
107
+ "loss": 9.75417251586914,
108
+ "step": 70
109
+ },
110
+ {
111
+ "epoch": 0.015,
112
+ "grad_norm": 0.6131484508514404,
113
+ "learning_rate": 1.1839999999999998e-06,
114
+ "loss": 9.632675170898438,
115
+ "step": 75
116
+ },
117
+ {
118
+ "epoch": 0.016,
119
+ "grad_norm": 0.605525553226471,
120
+ "learning_rate": 1.2639999999999999e-06,
121
+ "loss": 9.530663299560548,
122
+ "step": 80
123
+ },
124
+ {
125
+ "epoch": 0.017,
126
+ "grad_norm": 0.5865434408187866,
127
+ "learning_rate": 1.344e-06,
128
+ "loss": 9.533056640625,
129
+ "step": 85
130
+ },
131
+ {
132
+ "epoch": 0.018,
133
+ "grad_norm": 0.6370518207550049,
134
+ "learning_rate": 1.4239999999999998e-06,
135
+ "loss": 9.445852661132813,
136
+ "step": 90
137
+ },
138
+ {
139
+ "epoch": 0.019,
140
+ "grad_norm": 0.517352283000946,
141
+ "learning_rate": 1.504e-06,
142
+ "loss": 9.347330474853516,
143
+ "step": 95
144
+ },
145
+ {
146
+ "epoch": 0.02,
147
+ "grad_norm": 0.6384696960449219,
148
+ "learning_rate": 1.584e-06,
149
+ "loss": 9.231375122070313,
150
+ "step": 100
151
+ },
152
+ {
153
+ "epoch": 0.021,
154
+ "grad_norm": 0.5805168151855469,
155
+ "learning_rate": 1.6639999999999999e-06,
156
+ "loss": 9.230397796630859,
157
+ "step": 105
158
+ },
159
+ {
160
+ "epoch": 0.022,
161
+ "grad_norm": 0.5052227973937988,
162
+ "learning_rate": 1.744e-06,
163
+ "loss": 9.230642700195313,
164
+ "step": 110
165
+ },
166
+ {
167
+ "epoch": 0.023,
168
+ "grad_norm": 0.5524439811706543,
169
+ "learning_rate": 1.824e-06,
170
+ "loss": 9.179952239990234,
171
+ "step": 115
172
+ },
173
+ {
174
+ "epoch": 0.024,
175
+ "grad_norm": 0.5470523238182068,
176
+ "learning_rate": 1.904e-06,
177
+ "loss": 9.130237579345703,
178
+ "step": 120
179
+ },
180
+ {
181
+ "epoch": 0.025,
182
+ "grad_norm": 0.5232647657394409,
183
+ "learning_rate": 1.984e-06,
184
+ "loss": 9.082659912109374,
185
+ "step": 125
186
+ },
187
+ {
188
+ "epoch": 0.026,
189
+ "grad_norm": 0.47292640805244446,
190
+ "learning_rate": 2.064e-06,
191
+ "loss": 9.097735595703124,
192
+ "step": 130
193
+ },
194
+ {
195
+ "epoch": 0.027,
196
+ "grad_norm": 0.4722849726676941,
197
+ "learning_rate": 2.144e-06,
198
+ "loss": 9.046373748779297,
199
+ "step": 135
200
+ },
201
+ {
202
+ "epoch": 0.028,
203
+ "grad_norm": 0.4748806655406952,
204
+ "learning_rate": 2.2240000000000002e-06,
205
+ "loss": 9.014337921142578,
206
+ "step": 140
207
+ },
208
+ {
209
+ "epoch": 0.029,
210
+ "grad_norm": 0.4592997133731842,
211
+ "learning_rate": 2.304e-06,
212
+ "loss": 8.961915588378906,
213
+ "step": 145
214
+ },
215
+ {
216
+ "epoch": 0.03,
217
+ "grad_norm": 0.43399059772491455,
218
+ "learning_rate": 2.384e-06,
219
+ "loss": 8.965023040771484,
220
+ "step": 150
221
+ },
222
+ {
223
+ "epoch": 0.03,
224
+ "eval_accuracy": 0.10875213675213675,
225
+ "eval_loss": 8.931988716125488,
226
+ "eval_runtime": 10.0631,
227
+ "eval_samples_per_second": 9.937,
228
+ "eval_steps_per_second": 1.689,
229
+ "step": 150
230
+ },
231
+ {
232
+ "epoch": 0.031,
233
+ "grad_norm": 0.39372554421424866,
234
+ "learning_rate": 2.464e-06,
235
+ "loss": 8.943362426757812,
236
+ "step": 155
237
+ },
238
+ {
239
+ "epoch": 0.032,
240
+ "grad_norm": 0.4504595100879669,
241
+ "learning_rate": 2.544e-06,
242
+ "loss": 8.912525177001953,
243
+ "step": 160
244
+ },
245
+ {
246
+ "epoch": 0.033,
247
+ "grad_norm": 0.4033423960208893,
248
+ "learning_rate": 2.624e-06,
249
+ "loss": 8.891281127929688,
250
+ "step": 165
251
+ },
252
+ {
253
+ "epoch": 0.034,
254
+ "grad_norm": 0.4079058766365051,
255
+ "learning_rate": 2.704e-06,
256
+ "loss": 8.832948303222656,
257
+ "step": 170
258
+ },
259
+ {
260
+ "epoch": 0.035,
261
+ "grad_norm": 0.4341174066066742,
262
+ "learning_rate": 2.7839999999999995e-06,
263
+ "loss": 8.736992645263673,
264
+ "step": 175
265
+ },
266
+ {
267
+ "epoch": 0.036,
268
+ "grad_norm": 0.4039728045463562,
269
+ "learning_rate": 2.8639999999999996e-06,
270
+ "loss": 8.764207458496093,
271
+ "step": 180
272
+ },
273
+ {
274
+ "epoch": 0.037,
275
+ "grad_norm": 0.48787590861320496,
276
+ "learning_rate": 2.9439999999999997e-06,
277
+ "loss": 8.90152587890625,
278
+ "step": 185
279
+ },
280
+ {
281
+ "epoch": 0.038,
282
+ "grad_norm": 0.4344586730003357,
283
+ "learning_rate": 3.0239999999999998e-06,
284
+ "loss": 8.79043197631836,
285
+ "step": 190
286
+ },
287
+ {
288
+ "epoch": 0.039,
289
+ "grad_norm": 0.3761419653892517,
290
+ "learning_rate": 3.104e-06,
291
+ "loss": 8.768655395507812,
292
+ "step": 195
293
+ },
294
+ {
295
+ "epoch": 0.04,
296
+ "grad_norm": 0.40232253074645996,
297
+ "learning_rate": 3.184e-06,
298
+ "loss": 8.727652740478515,
299
+ "step": 200
300
+ },
301
+ {
302
+ "epoch": 0.041,
303
+ "grad_norm": 0.4565980136394501,
304
+ "learning_rate": 3.2639999999999996e-06,
305
+ "loss": 8.654769134521484,
306
+ "step": 205
307
+ },
308
+ {
309
+ "epoch": 0.042,
310
+ "grad_norm": 0.3786274492740631,
311
+ "learning_rate": 3.3439999999999997e-06,
312
+ "loss": 8.736286926269532,
313
+ "step": 210
314
+ },
315
+ {
316
+ "epoch": 0.043,
317
+ "grad_norm": 0.3933294415473938,
318
+ "learning_rate": 3.4239999999999997e-06,
319
+ "loss": 8.671070861816407,
320
+ "step": 215
321
+ },
322
+ {
323
+ "epoch": 0.044,
324
+ "grad_norm": 0.4195927083492279,
325
+ "learning_rate": 3.504e-06,
326
+ "loss": 8.66961898803711,
327
+ "step": 220
328
+ },
329
+ {
330
+ "epoch": 0.045,
331
+ "grad_norm": 0.3836154043674469,
332
+ "learning_rate": 3.584e-06,
333
+ "loss": 8.615521240234376,
334
+ "step": 225
335
+ },
336
+ {
337
+ "epoch": 0.046,
338
+ "grad_norm": 0.41303297877311707,
339
+ "learning_rate": 3.664e-06,
340
+ "loss": 8.627992248535156,
341
+ "step": 230
342
+ },
343
+ {
344
+ "epoch": 0.047,
345
+ "grad_norm": 0.4286525547504425,
346
+ "learning_rate": 3.744e-06,
347
+ "loss": 8.595711517333985,
348
+ "step": 235
349
+ },
350
+ {
351
+ "epoch": 0.048,
352
+ "grad_norm": 0.3865108788013458,
353
+ "learning_rate": 3.823999999999999e-06,
354
+ "loss": 8.616091918945312,
355
+ "step": 240
356
+ },
357
+ {
358
+ "epoch": 0.049,
359
+ "grad_norm": 0.4198671281337738,
360
+ "learning_rate": 3.903999999999999e-06,
361
+ "loss": 8.540158081054688,
362
+ "step": 245
363
+ },
364
+ {
365
+ "epoch": 0.05,
366
+ "grad_norm": 0.5092390179634094,
367
+ "learning_rate": 3.9839999999999995e-06,
368
+ "loss": 8.554955291748048,
369
+ "step": 250
370
+ },
371
+ {
372
+ "epoch": 0.051,
373
+ "grad_norm": 0.39753150939941406,
374
+ "learning_rate": 3.99999300106024e-06,
375
+ "loss": 8.531331634521484,
376
+ "step": 255
377
+ },
378
+ {
379
+ "epoch": 0.052,
380
+ "grad_norm": 0.382576584815979,
381
+ "learning_rate": 3.9999645679514235e-06,
382
+ "loss": 8.556974792480469,
383
+ "step": 260
384
+ },
385
+ {
386
+ "epoch": 0.053,
387
+ "grad_norm": 0.3579820394515991,
388
+ "learning_rate": 3.999914263550513e-06,
389
+ "loss": 8.572657012939453,
390
+ "step": 265
391
+ },
392
+ {
393
+ "epoch": 0.054,
394
+ "grad_norm": 0.6057927012443542,
395
+ "learning_rate": 3.9998420884076325e-06,
396
+ "loss": 8.486345672607422,
397
+ "step": 270
398
+ },
399
+ {
400
+ "epoch": 0.055,
401
+ "grad_norm": 0.3516131639480591,
402
+ "learning_rate": 3.999748043312075e-06,
403
+ "loss": 8.500395202636719,
404
+ "step": 275
405
+ },
406
+ {
407
+ "epoch": 0.056,
408
+ "grad_norm": 0.39441752433776855,
409
+ "learning_rate": 3.999632129292304e-06,
410
+ "loss": 8.484166717529297,
411
+ "step": 280
412
+ },
413
+ {
414
+ "epoch": 0.057,
415
+ "grad_norm": 0.42147520184516907,
416
+ "learning_rate": 3.9994943476159364e-06,
417
+ "loss": 8.458255004882812,
418
+ "step": 285
419
+ },
420
+ {
421
+ "epoch": 0.058,
422
+ "grad_norm": 0.3418014347553253,
423
+ "learning_rate": 3.999334699789731e-06,
424
+ "loss": 8.439933013916015,
425
+ "step": 290
426
+ },
427
+ {
428
+ "epoch": 0.059,
429
+ "grad_norm": 0.392968088388443,
430
+ "learning_rate": 3.99915318755957e-06,
431
+ "loss": 8.47509536743164,
432
+ "step": 295
433
+ },
434
+ {
435
+ "epoch": 0.06,
436
+ "grad_norm": 0.35503852367401123,
437
+ "learning_rate": 3.9989498129104425e-06,
438
+ "loss": 8.460848236083985,
439
+ "step": 300
440
+ },
441
+ {
442
+ "epoch": 0.06,
443
+ "eval_accuracy": 0.11944810744810745,
444
+ "eval_loss": 8.447799682617188,
445
+ "eval_runtime": 9.9265,
446
+ "eval_samples_per_second": 10.074,
447
+ "eval_steps_per_second": 1.713,
448
+ "step": 300
449
+ }
450
+ ],
451
+ "logging_steps": 5,
452
+ "max_steps": 5000,
453
+ "num_input_tokens_seen": 0,
454
+ "num_train_epochs": 9223372036854775807,
455
+ "save_steps": 300,
456
+ "stateful_callbacks": {
457
+ "TrainerControl": {
458
+ "args": {
459
+ "should_epoch_stop": false,
460
+ "should_evaluate": false,
461
+ "should_log": false,
462
+ "should_save": true,
463
+ "should_training_stop": false
464
+ },
465
+ "attributes": {}
466
+ }
467
+ },
468
+ "total_flos": 0.0,
469
+ "train_batch_size": 6,
470
+ "trial_name": null,
471
+ "trial_params": null
472
+ }
last-checkpoint/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:542c39b999873d04e8ee4a3271b21d58fcf1252ca48f9f04659e9b82e85ebaaa
3
+ size 5265