CodeIsAbstract commited on
Commit
ca9dac4
·
verified ·
1 Parent(s): e8bb815

Training in progress, step 6000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eec6ae3c232115c0c2fe4bcf646b1c03138328548b645022df7b55df2d6ff03b
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bcc177d71a1506005267437007f2de8d24b91691c13a50b732ea2bcf9fe2581f
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2c8dc5c30771f73079678db63765b219aafca9ea4b8e33ef2987d0d846fa62d3
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:acb87da978e2dad13433319692c27084c166b313f922269425e1c29be9ff602b
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7864fb6e928e45d4215a1ae96eab6e34cbc82f54e4a38e3cf2e1449f98239105
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7578bdc324f9e75df38586eb586132281169978ccdfb68af7c97dfe72a3c31b4
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bb8ff3d7c5e846455471242a9719dd9767ec3729631e3efca61eaecbd0ad4ee7
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ffeaed815dfe24a3d4d30e6928de6e80034c94a6e68899611f9c7a0a75684d5
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:561e0049aa6e5e793844e867537a9c947ef9ba1ad03fe6ba44d328d1dffc9702
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ead4bfc2097774b0bab8635cdb6eefc7edb67d58c75bf9f6514bf7d08d200842
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.26666666666666666,
6
  "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -324,6 +324,164 @@
324
  "eval_samples_per_second": 7.638,
325
  "eval_steps_per_second": 0.382,
326
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
327
  }
328
  ],
329
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4,
6
  "eval_steps": 1000,
7
+ "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
324
  "eval_samples_per_second": 7.638,
325
  "eval_steps_per_second": 0.382,
326
  "step": 4000
327
+ },
328
+ {
329
+ "epoch": 0.2733333333333333,
330
+ "grad_norm": 1.0034714937210083,
331
+ "learning_rate": 0.00034791880006519195,
332
+ "loss": 9.250709838867188,
333
+ "step": 4100
334
+ },
335
+ {
336
+ "epoch": 0.28,
337
+ "grad_norm": 1.2436392307281494,
338
+ "learning_rate": 0.00034491543277181,
339
+ "loss": 9.246346435546876,
340
+ "step": 4200
341
+ },
342
+ {
343
+ "epoch": 0.2866666666666667,
344
+ "grad_norm": 1.180080771446228,
345
+ "learning_rate": 0.00034184163395788343,
346
+ "loss": 9.188368530273438,
347
+ "step": 4300
348
+ },
349
+ {
350
+ "epoch": 0.29333333333333333,
351
+ "grad_norm": 1.0143563747406006,
352
+ "learning_rate": 0.00033869889754521314,
353
+ "loss": 9.10044677734375,
354
+ "step": 4400
355
+ },
356
+ {
357
+ "epoch": 0.3,
358
+ "grad_norm": 1.2389507293701172,
359
+ "learning_rate": 0.0003354887509605197,
360
+ "loss": 9.080850830078125,
361
+ "step": 4500
362
+ },
363
+ {
364
+ "epoch": 0.30666666666666664,
365
+ "grad_norm": 0.997056782245636,
366
+ "learning_rate": 0.0003322127543930859,
367
+ "loss": 9.041187744140625,
368
+ "step": 4600
369
+ },
370
+ {
371
+ "epoch": 0.31333333333333335,
372
+ "grad_norm": 1.006325364112854,
373
+ "learning_rate": 0.00032887250003647676,
374
+ "loss": 9.152498779296875,
375
+ "step": 4700
376
+ },
377
+ {
378
+ "epoch": 0.32,
379
+ "grad_norm": 1.1522862911224365,
380
+ "learning_rate": 0.00032546961131470485,
381
+ "loss": 8.979527587890624,
382
+ "step": 4800
383
+ },
384
+ {
385
+ "epoch": 0.32666666666666666,
386
+ "grad_norm": 0.9992244839668274,
387
+ "learning_rate": 0.00032200574209321657,
388
+ "loss": 8.930678100585938,
389
+ "step": 4900
390
+ },
391
+ {
392
+ "epoch": 0.3333333333333333,
393
+ "grad_norm": 1.188022255897522,
394
+ "learning_rate": 0.0003184825758750839,
395
+ "loss": 8.919720458984376,
396
+ "step": 5000
397
+ },
398
+ {
399
+ "epoch": 0.3333333333333333,
400
+ "eval_accuracy": 0.27488649706457924,
401
+ "eval_loss": 8.979753494262695,
402
+ "eval_runtime": 65.4227,
403
+ "eval_samples_per_second": 7.643,
404
+ "eval_steps_per_second": 0.382,
405
+ "step": 5000
406
+ },
407
+ {
408
+ "epoch": 0.34,
409
+ "grad_norm": 0.9783419966697693,
410
+ "learning_rate": 0.00031490182498279115,
411
+ "loss": 8.924981079101563,
412
+ "step": 5100
413
+ },
414
+ {
415
+ "epoch": 0.3466666666666667,
416
+ "grad_norm": 1.0431445837020874,
417
+ "learning_rate": 0.0003112652297260157,
418
+ "loss": 8.822501220703124,
419
+ "step": 5200
420
+ },
421
+ {
422
+ "epoch": 0.35333333333333333,
423
+ "grad_norm": 0.9755280613899231,
424
+ "learning_rate": 0.00030757455755580553,
425
+ "loss": 8.865919189453125,
426
+ "step": 5300
427
+ },
428
+ {
429
+ "epoch": 0.36,
430
+ "grad_norm": 1.1168733835220337,
431
+ "learning_rate": 0.0003038316022055665,
432
+ "loss": 8.899959716796875,
433
+ "step": 5400
434
+ },
435
+ {
436
+ "epoch": 0.36666666666666664,
437
+ "grad_norm": 0.9611853957176208,
438
+ "learning_rate": 0.00030003818281927526,
439
+ "loss": 8.851053466796875,
440
+ "step": 5500
441
+ },
442
+ {
443
+ "epoch": 0.37333333333333335,
444
+ "grad_norm": 1.040235996246338,
445
+ "learning_rate": 0.00029619614306734235,
446
+ "loss": 8.790068359375,
447
+ "step": 5600
448
+ },
449
+ {
450
+ "epoch": 0.38,
451
+ "grad_norm": 1.0123136043548584,
452
+ "learning_rate": 0.00029230735025055524,
453
+ "loss": 8.8330029296875,
454
+ "step": 5700
455
+ },
456
+ {
457
+ "epoch": 0.38666666666666666,
458
+ "grad_norm": 1.0311475992202759,
459
+ "learning_rate": 0.00028837369439253617,
460
+ "loss": 8.824832153320312,
461
+ "step": 5800
462
+ },
463
+ {
464
+ "epoch": 0.3933333333333333,
465
+ "grad_norm": 0.997328519821167,
466
+ "learning_rate": 0.0002843970873211566,
467
+ "loss": 8.828201904296876,
468
+ "step": 5900
469
+ },
470
+ {
471
+ "epoch": 0.4,
472
+ "grad_norm": 1.5238865613937378,
473
+ "learning_rate": 0.0002803794617393543,
474
+ "loss": 8.719149169921875,
475
+ "step": 6000
476
+ },
477
+ {
478
+ "epoch": 0.4,
479
+ "eval_accuracy": 0.28737377690802346,
480
+ "eval_loss": 8.731175422668457,
481
+ "eval_runtime": 65.6219,
482
+ "eval_samples_per_second": 7.619,
483
+ "eval_steps_per_second": 0.381,
484
+ "step": 6000
485
  }
486
  ],
487
  "logging_steps": 100,