CodeIsAbstract commited on
Commit
ca4feaf
·
verified ·
1 Parent(s): 51ddc0a

Training in progress, step 10000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5957954f25406e9df797421489d9b120ac7feff52692d5e5c70038c879f86412
3
  size 529337896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5380c87b8a1992262417a07584f7e2ad84f4c2f88005d931c232bf8ae2edd651
3
  size 529337896
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6ee0d0e9122e67a2d2d94315cf19cbd6a916fcb06556d38abe0ccc77fd7feae8
3
  size 871247243
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d345cb2206543c9c855d8cd93c9763b056b3b514641a19501f256c2c6ec77df7
3
  size 871247243
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dbcbab03d5bb70bb2a3066017b02040dc9b35585ddf112a4be38eb15410f67d9
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:58c4ef51004bd7cb63819242ee70337c219b896f88e964689c3e3e6e457044c6
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e9567d76a2efaa946d19913fd9fe6f6f70f6f7101d7dcb8866c09743534f2d4d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:833776bf99a0597a4bf38df86be38e8ac151736eba1d084c9cbb84cab1caf4e9
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.16,
6
  "eval_steps": 1000,
7
- "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -640,6 +640,164 @@
640
  "eval_samples_per_second": 56.787,
641
  "eval_steps_per_second": 0.454,
642
  "step": 8000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
643
  }
644
  ],
645
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2,
6
  "eval_steps": 1000,
7
+ "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
640
  "eval_samples_per_second": 56.787,
641
  "eval_steps_per_second": 0.454,
642
  "step": 8000
643
+ },
644
+ {
645
+ "epoch": 0.162,
646
+ "grad_norm": 0.13911907374858856,
647
+ "learning_rate": 0.0019322148017972016,
648
+ "loss": 3.646,
649
+ "step": 8100
650
+ },
651
+ {
652
+ "epoch": 0.164,
653
+ "grad_norm": 0.1297680288553238,
654
+ "learning_rate": 0.001929800831168135,
655
+ "loss": 3.6412,
656
+ "step": 8200
657
+ },
658
+ {
659
+ "epoch": 0.166,
660
+ "grad_norm": 0.14803367853164673,
661
+ "learning_rate": 0.001927346188038576,
662
+ "loss": 3.6298,
663
+ "step": 8300
664
+ },
665
+ {
666
+ "epoch": 0.168,
667
+ "grad_norm": 0.1325535923242569,
668
+ "learning_rate": 0.0019248509797825672,
669
+ "loss": 3.6067,
670
+ "step": 8400
671
+ },
672
+ {
673
+ "epoch": 0.17,
674
+ "grad_norm": 0.16948860883712769,
675
+ "learning_rate": 0.0019223153155486009,
676
+ "loss": 3.6284,
677
+ "step": 8500
678
+ },
679
+ {
680
+ "epoch": 0.172,
681
+ "grad_norm": 0.12973898649215698,
682
+ "learning_rate": 0.0019197393062548454,
683
+ "loss": 3.627,
684
+ "step": 8600
685
+ },
686
+ {
687
+ "epoch": 0.174,
688
+ "grad_norm": 0.1576964110136032,
689
+ "learning_rate": 0.0019171230645842923,
690
+ "loss": 3.6027,
691
+ "step": 8700
692
+ },
693
+ {
694
+ "epoch": 0.176,
695
+ "grad_norm": 0.15394999086856842,
696
+ "learning_rate": 0.0019144667049798272,
697
+ "loss": 3.6098,
698
+ "step": 8800
699
+ },
700
+ {
701
+ "epoch": 0.178,
702
+ "grad_norm": 0.12594962120056152,
703
+ "learning_rate": 0.0019117703436392253,
704
+ "loss": 3.6221,
705
+ "step": 8900
706
+ },
707
+ {
708
+ "epoch": 0.18,
709
+ "grad_norm": 0.1420104056596756,
710
+ "learning_rate": 0.001909034098510066,
711
+ "loss": 3.5993,
712
+ "step": 9000
713
+ },
714
+ {
715
+ "epoch": 0.18,
716
+ "eval_accuracy": 0.3549628180039139,
717
+ "eval_loss": 3.570964813232422,
718
+ "eval_runtime": 12.4303,
719
+ "eval_samples_per_second": 80.449,
720
+ "eval_steps_per_second": 0.644,
721
+ "step": 9000
722
+ },
723
+ {
724
+ "epoch": 0.182,
725
+ "grad_norm": 0.12720872461795807,
726
+ "learning_rate": 0.001906258089284576,
727
+ "loss": 3.5913,
728
+ "step": 9100
729
+ },
730
+ {
731
+ "epoch": 0.184,
732
+ "grad_norm": 0.14890190958976746,
733
+ "learning_rate": 0.0019034424373943915,
734
+ "loss": 3.6113,
735
+ "step": 9200
736
+ },
737
+ {
738
+ "epoch": 0.186,
739
+ "grad_norm": 0.15023267269134521,
740
+ "learning_rate": 0.0019005872660052478,
741
+ "loss": 3.6011,
742
+ "step": 9300
743
+ },
744
+ {
745
+ "epoch": 0.188,
746
+ "grad_norm": 0.1637164056301117,
747
+ "learning_rate": 0.001897692700011591,
748
+ "loss": 3.5932,
749
+ "step": 9400
750
+ },
751
+ {
752
+ "epoch": 0.19,
753
+ "grad_norm": 0.13102982938289642,
754
+ "learning_rate": 0.0018947588660311143,
755
+ "loss": 3.5889,
756
+ "step": 9500
757
+ },
758
+ {
759
+ "epoch": 0.192,
760
+ "grad_norm": 0.16267850995063782,
761
+ "learning_rate": 0.0018917858923992211,
762
+ "loss": 3.5939,
763
+ "step": 9600
764
+ },
765
+ {
766
+ "epoch": 0.194,
767
+ "grad_norm": 0.12798982858657837,
768
+ "learning_rate": 0.0018887739091634085,
769
+ "loss": 3.5896,
770
+ "step": 9700
771
+ },
772
+ {
773
+ "epoch": 0.196,
774
+ "grad_norm": 0.13872383534908295,
775
+ "learning_rate": 0.0018857230480775807,
776
+ "loss": 3.574,
777
+ "step": 9800
778
+ },
779
+ {
780
+ "epoch": 0.198,
781
+ "grad_norm": 0.11993825435638428,
782
+ "learning_rate": 0.0018826334425962855,
783
+ "loss": 3.589,
784
+ "step": 9900
785
+ },
786
+ {
787
+ "epoch": 0.2,
788
+ "grad_norm": 0.13074541091918945,
789
+ "learning_rate": 0.0018795052278688753,
790
+ "loss": 3.5828,
791
+ "step": 10000
792
+ },
793
+ {
794
+ "epoch": 0.2,
795
+ "eval_accuracy": 0.35816046966731896,
796
+ "eval_loss": 3.540745973587036,
797
+ "eval_runtime": 17.8387,
798
+ "eval_samples_per_second": 56.058,
799
+ "eval_steps_per_second": 0.448,
800
+ "step": 10000
801
  }
802
  ],
803
  "logging_steps": 100,