CodeIsAbstract commited on
Commit
be06d86
·
verified ·
1 Parent(s): 537ea27

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bcc177d71a1506005267437007f2de8d24b91691c13a50b732ea2bcf9fe2581f
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e4640cdfd5049fa55f638b28d54cd06e5001e7d6ec0e00957bc1d8edde2a495a
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:acb87da978e2dad13433319692c27084c166b313f922269425e1c29be9ff602b
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:64a56a8a2bc15f4d758b0eb9991e9912dcd8f3557ca8a359e579b699adf35587
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7578bdc324f9e75df38586eb586132281169978ccdfb68af7c97dfe72a3c31b4
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af7bb138aa5106db6be0a3fa0ca8b584ca968cb624961c21434f43b0f94a7f53
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6ffeaed815dfe24a3d4d30e6928de6e80034c94a6e68899611f9c7a0a75684d5
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:53d5b14b9a348d0c0d9daa4cabaaf2992e8a1a55a0f722d47cf938ca82885ab9
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ead4bfc2097774b0bab8635cdb6eefc7edb67d58c75bf9f6514bf7d08d200842
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3dff6d0a97e77162b13dd1d33a894b8edaf14cd2511685238f1583301c4a8320
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -482,6 +482,164 @@
482
  "eval_samples_per_second": 7.619,
483
  "eval_steps_per_second": 0.381,
484
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
485
  }
486
  ],
487
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.5333333333333333,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
482
  "eval_samples_per_second": 7.619,
483
  "eval_steps_per_second": 0.381,
484
  "step": 6000
485
+ },
486
+ {
487
+ "epoch": 0.4066666666666667,
488
+ "grad_norm": 1.1989500522613525,
489
+ "learning_rate": 0.0002763227702858047,
490
+ "loss": 8.64713623046875,
491
+ "step": 6100
492
+ },
493
+ {
494
+ "epoch": 0.41333333333333333,
495
+ "grad_norm": 1.3336397409439087,
496
+ "learning_rate": 0.00027222898458590343,
497
+ "loss": 8.749034423828125,
498
+ "step": 6200
499
+ },
500
+ {
501
+ "epoch": 0.42,
502
+ "grad_norm": 1.0257115364074707,
503
+ "learning_rate": 0.0002681000942935204,
504
+ "loss": 8.847327880859375,
505
+ "step": 6300
506
+ },
507
+ {
508
+ "epoch": 0.4266666666666667,
509
+ "grad_norm": 1.03694748878479,
510
+ "learning_rate": 0.0002639381061239921,
511
+ "loss": 8.704371337890626,
512
+ "step": 6400
513
+ },
514
+ {
515
+ "epoch": 0.43333333333333335,
516
+ "grad_norm": 0.9518272876739502,
517
+ "learning_rate": 0.00025974504287882194,
518
+ "loss": 8.664287109375,
519
+ "step": 6500
520
+ },
521
+ {
522
+ "epoch": 0.44,
523
+ "grad_norm": 1.0655875205993652,
524
+ "learning_rate": 0.000255522942462562,
525
+ "loss": 8.662864990234375,
526
+ "step": 6600
527
+ },
528
+ {
529
+ "epoch": 0.44666666666666666,
530
+ "grad_norm": 1.0382905006408691,
531
+ "learning_rate": 0.00025127385689235426,
532
+ "loss": 8.6200341796875,
533
+ "step": 6700
534
+ },
535
+ {
536
+ "epoch": 0.4533333333333333,
537
+ "grad_norm": 1.3670793771743774,
538
+ "learning_rate": 0.00024699985130061374,
539
+ "loss": 8.617677612304687,
540
+ "step": 6800
541
+ },
542
+ {
543
+ "epoch": 0.46,
544
+ "grad_norm": 1.2171597480773926,
545
+ "learning_rate": 0.0002427030029313362,
546
+ "loss": 8.605474853515625,
547
+ "step": 6900
548
+ },
549
+ {
550
+ "epoch": 0.4666666666666667,
551
+ "grad_norm": 1.1724241971969604,
552
+ "learning_rate": 0.00023838540013052062,
553
+ "loss": 8.558206787109375,
554
+ "step": 7000
555
+ },
556
+ {
557
+ "epoch": 0.4666666666666667,
558
+ "eval_accuracy": 0.2943894324853229,
559
+ "eval_loss": 8.570584297180176,
560
+ "eval_runtime": 65.4101,
561
+ "eval_samples_per_second": 7.644,
562
+ "eval_steps_per_second": 0.382,
563
+ "step": 7000
564
+ },
565
+ {
566
+ "epoch": 0.47333333333333333,
567
+ "grad_norm": 1.0969085693359375,
568
+ "learning_rate": 0.00023404914133119486,
569
+ "loss": 8.515238037109375,
570
+ "step": 7100
571
+ },
572
+ {
573
+ "epoch": 0.48,
574
+ "grad_norm": 1.4420628547668457,
575
+ "learning_rate": 0.00022969633403353913,
576
+ "loss": 8.62325927734375,
577
+ "step": 7200
578
+ },
579
+ {
580
+ "epoch": 0.4866666666666667,
581
+ "grad_norm": 1.0490083694458008,
582
+ "learning_rate": 0.0002253290937806034,
583
+ "loss": 8.469694213867188,
584
+ "step": 7300
585
+ },
586
+ {
587
+ "epoch": 0.49333333333333335,
588
+ "grad_norm": 1.3232401609420776,
589
+ "learning_rate": 0.00022094954313011468,
590
+ "loss": 8.4819580078125,
591
+ "step": 7400
592
+ },
593
+ {
594
+ "epoch": 0.5,
595
+ "grad_norm": 1.0941740274429321,
596
+ "learning_rate": 0.0002165598106228758,
597
+ "loss": 8.58266357421875,
598
+ "step": 7500
599
+ },
600
+ {
601
+ "epoch": 0.5066666666666667,
602
+ "grad_norm": 1.1495511531829834,
603
+ "learning_rate": 0.00021216202974825614,
604
+ "loss": 8.533701171875,
605
+ "step": 7600
606
+ },
607
+ {
608
+ "epoch": 0.5133333333333333,
609
+ "grad_norm": 1.004111886024475,
610
+ "learning_rate": 0.00020775833790727695,
611
+ "loss": 8.46027099609375,
612
+ "step": 7700
613
+ },
614
+ {
615
+ "epoch": 0.52,
616
+ "grad_norm": 1.4242042303085327,
617
+ "learning_rate": 0.0002033508753737963,
618
+ "loss": 8.484051513671876,
619
+ "step": 7800
620
+ },
621
+ {
622
+ "epoch": 0.5266666666666666,
623
+ "grad_norm": 1.1219866275787354,
624
+ "learning_rate": 0.00019894178425429674,
625
+ "loss": 8.434028930664063,
626
+ "step": 7900
627
+ },
628
+ {
629
+ "epoch": 0.5333333333333333,
630
+ "grad_norm": 1.009364128112793,
631
+ "learning_rate": 0.00019453320744678324,
632
+ "loss": 8.4456640625,
633
+ "step": 8000
634
+ },
635
+ {
636
+ "epoch": 0.5333333333333333,
637
+ "eval_accuracy": 0.29964774951076323,
638
+ "eval_loss": 8.433549880981445,
639
+ "eval_runtime": 65.652,
640
+ "eval_samples_per_second": 7.616,
641
+ "eval_steps_per_second": 0.381,
642
+ "step": 8000
643
  }
644
  ],
645
  "logging_steps": 100,