CodeIsAbstract commited on
Commit
0018a67
·
verified ·
1 Parent(s): 3492803

Training in progress, step 600, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f2a195bb2affd8b92355b3c3fbcdcbd8a8755d76f59c56edd91c1c227737ee85
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fec3e662918181e81fec2617a5f9f5de6a01b3e7022f265f73654bbc50438347
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c3e72953437bdac655e896c21e48e0325851505cdf5c9e67ff62fe154ccbd5a6
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38ffba74dde1d2bade94cf7d477b69f097fefa5eddd6cbfa453ed04630627cc1
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7b33358d13198554488bc16b3454035cd57e8325ffb080d0773f728d9e98a2a6
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af6ae550ab924258b75cba7b10f460b254db57e2f0ee1e6652ad45c4ae51f55f
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4be3090ef898425f736ce392d528a5e318dff0385f66de485a25e7ee90ff9432
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e216a9e826d4c344530b7e6d00074e0b36744d381337d9fe3838255770b48b79
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3b622a9357894e5e2498db0562bf753e1b98ac7c3a28c1715b2965db7bdcc60b
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:303505cecb97596e2f53b5d89a1eed07f741f9537510242aaeccaa44a51fbf19
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.14285714285714285,
6
  "eval_steps": 150,
7
- "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -236,6 +236,234 @@
236
  "eval_samples_per_second": 19.237,
237
  "eval_steps_per_second": 2.424,
238
  "step": 300
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
239
  }
240
  ],
241
  "logging_steps": 10,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2857142857142857,
6
  "eval_steps": 150,
7
+ "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
236
  "eval_samples_per_second": 19.237,
237
  "eval_steps_per_second": 2.424,
238
  "step": 300
239
+ },
240
+ {
241
+ "epoch": 0.14761904761904762,
242
+ "grad_norm": 59.79628372192383,
243
+ "learning_rate": 3.897685737763927e-06,
244
+ "loss": 21.495401000976564,
245
+ "step": 310
246
+ },
247
+ {
248
+ "epoch": 0.1523809523809524,
249
+ "grad_norm": 52.11283874511719,
250
+ "learning_rate": 3.887506463638843e-06,
251
+ "loss": 21.374043273925782,
252
+ "step": 320
253
+ },
254
+ {
255
+ "epoch": 0.15714285714285714,
256
+ "grad_norm": 17.035280227661133,
257
+ "learning_rate": 3.876859138254304e-06,
258
+ "loss": 20.709390258789064,
259
+ "step": 330
260
+ },
261
+ {
262
+ "epoch": 0.1619047619047619,
263
+ "grad_norm": 47.00483322143555,
264
+ "learning_rate": 3.865746401863021e-06,
265
+ "loss": 20.654705810546876,
266
+ "step": 340
267
+ },
268
+ {
269
+ "epoch": 0.16666666666666666,
270
+ "grad_norm": 47.759559631347656,
271
+ "learning_rate": 3.854171010127219e-06,
272
+ "loss": 20.17596435546875,
273
+ "step": 350
274
+ },
275
+ {
276
+ "epoch": 0.17142857142857143,
277
+ "grad_norm": 47.72419738769531,
278
+ "learning_rate": 3.842135833435311e-06,
279
+ "loss": 20.0145751953125,
280
+ "step": 360
281
+ },
282
+ {
283
+ "epoch": 0.1761904761904762,
284
+ "grad_norm": 51.43965148925781,
285
+ "learning_rate": 3.829643856190115e-06,
286
+ "loss": 20.162620544433594,
287
+ "step": 370
288
+ },
289
+ {
290
+ "epoch": 0.18095238095238095,
291
+ "grad_norm": 77.80012512207031,
292
+ "learning_rate": 3.8166981760688015e-06,
293
+ "loss": 19.78917694091797,
294
+ "step": 380
295
+ },
296
+ {
297
+ "epoch": 0.18571428571428572,
298
+ "grad_norm": 51.120635986328125,
299
+ "learning_rate": 3.8033020032547535e-06,
300
+ "loss": 19.7965087890625,
301
+ "step": 390
302
+ },
303
+ {
304
+ "epoch": 0.19047619047619047,
305
+ "grad_norm": 35.47827911376953,
306
+ "learning_rate": 3.7894586596415266e-06,
307
+ "loss": 19.693804931640624,
308
+ "step": 400
309
+ },
310
+ {
311
+ "epoch": 0.19523809523809524,
312
+ "grad_norm": 36.45315933227539,
313
+ "learning_rate": 3.775171578009108e-06,
314
+ "loss": 19.68980255126953,
315
+ "step": 410
316
+ },
317
+ {
318
+ "epoch": 0.2,
319
+ "grad_norm": 19.75481605529785,
320
+ "learning_rate": 3.7604443011726775e-06,
321
+ "loss": 19.586700439453125,
322
+ "step": 420
323
+ },
324
+ {
325
+ "epoch": 0.20476190476190476,
326
+ "grad_norm": 61.62051773071289,
327
+ "learning_rate": 3.7452804811040844e-06,
328
+ "loss": 19.429611206054688,
329
+ "step": 430
330
+ },
331
+ {
332
+ "epoch": 0.20952380952380953,
333
+ "grad_norm": 20.707780838012695,
334
+ "learning_rate": 3.7296838780262574e-06,
335
+ "loss": 19.41503143310547,
336
+ "step": 440
337
+ },
338
+ {
339
+ "epoch": 0.21428571428571427,
340
+ "grad_norm": 61.07346725463867,
341
+ "learning_rate": 3.713658359480768e-06,
342
+ "loss": 19.5462646484375,
343
+ "step": 450
344
+ },
345
+ {
346
+ "epoch": 0.21428571428571427,
347
+ "eval_accuracy": 0.048236220472440944,
348
+ "eval_loss": 18.801279067993164,
349
+ "eval_runtime": 25.633,
350
+ "eval_samples_per_second": 19.506,
351
+ "eval_steps_per_second": 2.458,
352
+ "step": 450
353
+ },
354
+ {
355
+ "epoch": 0.21904761904761905,
356
+ "grad_norm": 20.04033088684082,
357
+ "learning_rate": 3.6972078993687815e-06,
358
+ "loss": 18.968704223632812,
359
+ "step": 460
360
+ },
361
+ {
362
+ "epoch": 0.22380952380952382,
363
+ "grad_norm": 65.65965270996094,
364
+ "learning_rate": 3.680336576965642e-06,
365
+ "loss": 18.938497924804686,
366
+ "step": 470
367
+ },
368
+ {
369
+ "epoch": 0.22857142857142856,
370
+ "grad_norm": 48.90804672241211,
371
+ "learning_rate": 3.663048575909311e-06,
372
+ "loss": 18.844175720214842,
373
+ "step": 480
374
+ },
375
+ {
376
+ "epoch": 0.23333333333333334,
377
+ "grad_norm": 17.749887466430664,
378
+ "learning_rate": 3.645348183162947e-06,
379
+ "loss": 18.766136169433594,
380
+ "step": 490
381
+ },
382
+ {
383
+ "epoch": 0.23809523809523808,
384
+ "grad_norm": 43.13874435424805,
385
+ "learning_rate": 3.627239787951847e-06,
386
+ "loss": 18.34910888671875,
387
+ "step": 500
388
+ },
389
+ {
390
+ "epoch": 0.24285714285714285,
391
+ "grad_norm": 29.883098602294922,
392
+ "learning_rate": 3.608727880675036e-06,
393
+ "loss": 18.39371337890625,
394
+ "step": 510
395
+ },
396
+ {
397
+ "epoch": 0.24761904761904763,
398
+ "grad_norm": 23.2874698638916,
399
+ "learning_rate": 3.58981705179177e-06,
400
+ "loss": 18.52344207763672,
401
+ "step": 520
402
+ },
403
+ {
404
+ "epoch": 0.2523809523809524,
405
+ "grad_norm": 28.679786682128906,
406
+ "learning_rate": 3.570511990683222e-06,
407
+ "loss": 18.389620971679687,
408
+ "step": 530
409
+ },
410
+ {
411
+ "epoch": 0.2571428571428571,
412
+ "grad_norm": 39.36701965332031,
413
+ "learning_rate": 3.550817484489643e-06,
414
+ "loss": 18.337835693359374,
415
+ "step": 540
416
+ },
417
+ {
418
+ "epoch": 0.2619047619047619,
419
+ "grad_norm": 14.972796440124512,
420
+ "learning_rate": 3.5307384169232777e-06,
421
+ "loss": 18.569638061523438,
422
+ "step": 550
423
+ },
424
+ {
425
+ "epoch": 0.26666666666666666,
426
+ "grad_norm": 20.092327117919922,
427
+ "learning_rate": 3.5102797670573345e-06,
428
+ "loss": 18.32196044921875,
429
+ "step": 560
430
+ },
431
+ {
432
+ "epoch": 0.2714285714285714,
433
+ "grad_norm": 42.83948516845703,
434
+ "learning_rate": 3.4894466080913077e-06,
435
+ "loss": 18.12080078125,
436
+ "step": 570
437
+ },
438
+ {
439
+ "epoch": 0.2761904761904762,
440
+ "grad_norm": 150.76181030273438,
441
+ "learning_rate": 3.4682441060929587e-06,
442
+ "loss": 18.27479248046875,
443
+ "step": 580
444
+ },
445
+ {
446
+ "epoch": 0.28095238095238095,
447
+ "grad_norm": 24.40485382080078,
448
+ "learning_rate": 3.446677518717271e-06,
449
+ "loss": 18.172178649902342,
450
+ "step": 590
451
+ },
452
+ {
453
+ "epoch": 0.2857142857142857,
454
+ "grad_norm": 21.843042373657227,
455
+ "learning_rate": 3.4247521939026897e-06,
456
+ "loss": 17.919189453125,
457
+ "step": 600
458
+ },
459
+ {
460
+ "epoch": 0.2857142857142857,
461
+ "eval_accuracy": 0.05248031496062992,
462
+ "eval_loss": 17.725130081176758,
463
+ "eval_runtime": 25.3949,
464
+ "eval_samples_per_second": 19.689,
465
+ "eval_steps_per_second": 2.481,
466
+ "step": 600
467
  }
468
  ],
469
  "logging_steps": 10,