CodeIsAbstract commited on
Commit
355816b
·
verified ·
1 Parent(s): baaba00

Training in progress, step 600, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fdfa536891b8ffc83f63e9ace650cf6e9f915075b64c129dc410645e45a1688a
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad916a9d642533dedc2b97aa7773d01801ef186b4e50812f6dd96caf10edd3c9
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ba78434b69e3d278adef8b3e6500747e43a121a034e99f8a3ff4d1bde8a22b60
3
  size 350603
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:03fffafe35924680926a93c156f62f86e42af54dff05c03b31aa9af53da4d58d
3
  size 350603
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9079206c3c25e14fa2f8a19a24a32c2e5380266de4ccca26cb728acd92fae5e8
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b6bed560ba82ad938724499dcd6b1144659f41214cf5bc1f5da6347e162b25
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c18d3014a31b7d0b10d3631683a1f76c56eca77072280718fe48627ba3a1c6f4
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca2b96950726cb5d1ecc220f8fe4a4ed97580d0d059753b7132b783c68be6714
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.06,
6
  "eval_steps": 150,
7
- "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -446,6 +446,444 @@
446
  "eval_samples_per_second": 9.591,
447
  "eval_steps_per_second": 1.631,
448
  "step": 300
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
449
  }
450
  ],
451
  "logging_steps": 5,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.12,
6
  "eval_steps": 150,
7
+ "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
446
  "eval_samples_per_second": 9.591,
447
  "eval_steps_per_second": 1.631,
448
  "step": 300
449
+ },
450
+ {
451
+ "epoch": 0.061,
452
+ "grad_norm": 0.0692087933421135,
453
+ "learning_rate": 3.998724578066421e-06,
454
+ "loss": 7.243695831298828,
455
+ "step": 305
456
+ },
457
+ {
458
+ "epoch": 0.062,
459
+ "grad_norm": 0.08431071788072586,
460
+ "learning_rate": 3.998477485490638e-06,
461
+ "loss": 7.292684173583984,
462
+ "step": 310
463
+ },
464
+ {
465
+ "epoch": 0.063,
466
+ "grad_norm": 0.08417057991027832,
467
+ "learning_rate": 3.998208537885259e-06,
468
+ "loss": 7.222378540039062,
469
+ "step": 315
470
+ },
471
+ {
472
+ "epoch": 0.064,
473
+ "grad_norm": 0.07904339581727982,
474
+ "learning_rate": 3.997917738191449e-06,
475
+ "loss": 7.278653717041015,
476
+ "step": 320
477
+ },
478
+ {
479
+ "epoch": 0.065,
480
+ "grad_norm": 0.13258706033229828,
481
+ "learning_rate": 3.99760508958935e-06,
482
+ "loss": 7.202552795410156,
483
+ "step": 325
484
+ },
485
+ {
486
+ "epoch": 0.066,
487
+ "grad_norm": 0.0901857390999794,
488
+ "learning_rate": 3.997270595498036e-06,
489
+ "loss": 7.205686950683594,
490
+ "step": 330
491
+ },
492
+ {
493
+ "epoch": 0.067,
494
+ "grad_norm": 0.07317949831485748,
495
+ "learning_rate": 3.99691425957548e-06,
496
+ "loss": 7.293389892578125,
497
+ "step": 335
498
+ },
499
+ {
500
+ "epoch": 0.068,
501
+ "grad_norm": 0.11255510151386261,
502
+ "learning_rate": 3.996536085718516e-06,
503
+ "loss": 7.229725646972656,
504
+ "step": 340
505
+ },
506
+ {
507
+ "epoch": 0.069,
508
+ "grad_norm": 0.07706659287214279,
509
+ "learning_rate": 3.996136078062792e-06,
510
+ "loss": 7.257555389404297,
511
+ "step": 345
512
+ },
513
+ {
514
+ "epoch": 0.07,
515
+ "grad_norm": 0.08863229304552078,
516
+ "learning_rate": 3.995714240982727e-06,
517
+ "loss": 7.257266235351563,
518
+ "step": 350
519
+ },
520
+ {
521
+ "epoch": 0.071,
522
+ "grad_norm": 0.08663391321897507,
523
+ "learning_rate": 3.995270579091465e-06,
524
+ "loss": 7.118687438964844,
525
+ "step": 355
526
+ },
527
+ {
528
+ "epoch": 0.072,
529
+ "grad_norm": 0.08917262405157089,
530
+ "learning_rate": 3.99480509724082e-06,
531
+ "loss": 7.238731384277344,
532
+ "step": 360
533
+ },
534
+ {
535
+ "epoch": 0.073,
536
+ "grad_norm": 0.09512999653816223,
537
+ "learning_rate": 3.994317800521228e-06,
538
+ "loss": 7.2104545593261715,
539
+ "step": 365
540
+ },
541
+ {
542
+ "epoch": 0.074,
543
+ "grad_norm": 0.10050003230571747,
544
+ "learning_rate": 3.993808694261687e-06,
545
+ "loss": 7.186836242675781,
546
+ "step": 370
547
+ },
548
+ {
549
+ "epoch": 0.075,
550
+ "grad_norm": 0.12758338451385498,
551
+ "learning_rate": 3.993277784029702e-06,
552
+ "loss": 7.171599578857422,
553
+ "step": 375
554
+ },
555
+ {
556
+ "epoch": 0.076,
557
+ "grad_norm": 0.08614124357700348,
558
+ "learning_rate": 3.992725075631223e-06,
559
+ "loss": 7.201605224609375,
560
+ "step": 380
561
+ },
562
+ {
563
+ "epoch": 0.077,
564
+ "grad_norm": 0.07332151383161545,
565
+ "learning_rate": 3.992150575110579e-06,
566
+ "loss": 7.026101684570312,
567
+ "step": 385
568
+ },
569
+ {
570
+ "epoch": 0.078,
571
+ "grad_norm": 0.0854375809431076,
572
+ "learning_rate": 3.991554288750416e-06,
573
+ "loss": 7.223291015625,
574
+ "step": 390
575
+ },
576
+ {
577
+ "epoch": 0.079,
578
+ "grad_norm": 0.09470361471176147,
579
+ "learning_rate": 3.990936223071627e-06,
580
+ "loss": 7.0966438293457035,
581
+ "step": 395
582
+ },
583
+ {
584
+ "epoch": 0.08,
585
+ "grad_norm": 0.06626710295677185,
586
+ "learning_rate": 3.990296384833279e-06,
587
+ "loss": 7.142760467529297,
588
+ "step": 400
589
+ },
590
+ {
591
+ "epoch": 0.081,
592
+ "grad_norm": 0.13561668992042542,
593
+ "learning_rate": 3.989634781032539e-06,
594
+ "loss": 7.0024574279785154,
595
+ "step": 405
596
+ },
597
+ {
598
+ "epoch": 0.082,
599
+ "grad_norm": 0.09113086760044098,
600
+ "learning_rate": 3.9889514189046015e-06,
601
+ "loss": 7.137001800537109,
602
+ "step": 410
603
+ },
604
+ {
605
+ "epoch": 0.083,
606
+ "grad_norm": 0.07085978239774704,
607
+ "learning_rate": 3.988246305922606e-06,
608
+ "loss": 7.029933929443359,
609
+ "step": 415
610
+ },
611
+ {
612
+ "epoch": 0.084,
613
+ "grad_norm": 0.06468215584754944,
614
+ "learning_rate": 3.987519449797554e-06,
615
+ "loss": 7.02655029296875,
616
+ "step": 420
617
+ },
618
+ {
619
+ "epoch": 0.085,
620
+ "grad_norm": 0.12091367691755295,
621
+ "learning_rate": 3.986770858478227e-06,
622
+ "loss": 7.116294097900391,
623
+ "step": 425
624
+ },
625
+ {
626
+ "epoch": 0.086,
627
+ "grad_norm": 0.10934042185544968,
628
+ "learning_rate": 3.986000540151101e-06,
629
+ "loss": 7.093238067626953,
630
+ "step": 430
631
+ },
632
+ {
633
+ "epoch": 0.087,
634
+ "grad_norm": 0.07274443656206131,
635
+ "learning_rate": 3.985208503240253e-06,
636
+ "loss": 7.123175048828125,
637
+ "step": 435
638
+ },
639
+ {
640
+ "epoch": 0.088,
641
+ "grad_norm": 0.0802014172077179,
642
+ "learning_rate": 3.9843947564072725e-06,
643
+ "loss": 7.188668060302734,
644
+ "step": 440
645
+ },
646
+ {
647
+ "epoch": 0.089,
648
+ "grad_norm": 0.06854639202356339,
649
+ "learning_rate": 3.983559308551164e-06,
650
+ "loss": 7.106547546386719,
651
+ "step": 445
652
+ },
653
+ {
654
+ "epoch": 0.09,
655
+ "grad_norm": 0.07703255116939545,
656
+ "learning_rate": 3.982702168808252e-06,
657
+ "loss": 7.064980316162109,
658
+ "step": 450
659
+ },
660
+ {
661
+ "epoch": 0.09,
662
+ "eval_accuracy": 0.12983638583638585,
663
+ "eval_loss": 6.963926315307617,
664
+ "eval_runtime": 10.3988,
665
+ "eval_samples_per_second": 9.616,
666
+ "eval_steps_per_second": 1.635,
667
+ "step": 450
668
+ },
669
+ {
670
+ "epoch": 0.091,
671
+ "grad_norm": 0.08548009395599365,
672
+ "learning_rate": 3.981823346552078e-06,
673
+ "loss": 7.028290557861328,
674
+ "step": 455
675
+ },
676
+ {
677
+ "epoch": 0.092,
678
+ "grad_norm": 0.08591306209564209,
679
+ "learning_rate": 3.980922851393302e-06,
680
+ "loss": 7.01605453491211,
681
+ "step": 460
682
+ },
683
+ {
684
+ "epoch": 0.093,
685
+ "grad_norm": 0.07920879870653152,
686
+ "learning_rate": 3.9800006931795954e-06,
687
+ "loss": 6.990329742431641,
688
+ "step": 465
689
+ },
690
+ {
691
+ "epoch": 0.094,
692
+ "grad_norm": 0.07192882895469666,
693
+ "learning_rate": 3.979056881995533e-06,
694
+ "loss": 6.953098297119141,
695
+ "step": 470
696
+ },
697
+ {
698
+ "epoch": 0.095,
699
+ "grad_norm": 0.06649420410394669,
700
+ "learning_rate": 3.978091428162482e-06,
701
+ "loss": 6.951087951660156,
702
+ "step": 475
703
+ },
704
+ {
705
+ "epoch": 0.096,
706
+ "grad_norm": 0.08228413760662079,
707
+ "learning_rate": 3.97710434223849e-06,
708
+ "loss": 7.034120178222656,
709
+ "step": 480
710
+ },
711
+ {
712
+ "epoch": 0.097,
713
+ "grad_norm": 0.10376594960689545,
714
+ "learning_rate": 3.976095635018172e-06,
715
+ "loss": 6.995967864990234,
716
+ "step": 485
717
+ },
718
+ {
719
+ "epoch": 0.098,
720
+ "grad_norm": 0.06993073225021362,
721
+ "learning_rate": 3.975065317532588e-06,
722
+ "loss": 6.952136993408203,
723
+ "step": 490
724
+ },
725
+ {
726
+ "epoch": 0.099,
727
+ "grad_norm": 0.08720093965530396,
728
+ "learning_rate": 3.974013401049125e-06,
729
+ "loss": 6.902699279785156,
730
+ "step": 495
731
+ },
732
+ {
733
+ "epoch": 0.1,
734
+ "grad_norm": 0.1016659289598465,
735
+ "learning_rate": 3.972939897071373e-06,
736
+ "loss": 6.927466583251953,
737
+ "step": 500
738
+ },
739
+ {
740
+ "epoch": 0.101,
741
+ "grad_norm": 0.06847221404314041,
742
+ "learning_rate": 3.971844817338999e-06,
743
+ "loss": 6.902294921875,
744
+ "step": 505
745
+ },
746
+ {
747
+ "epoch": 0.102,
748
+ "grad_norm": 0.07875518500804901,
749
+ "learning_rate": 3.97072817382762e-06,
750
+ "loss": 6.962755584716797,
751
+ "step": 510
752
+ },
753
+ {
754
+ "epoch": 0.103,
755
+ "grad_norm": 0.08157093822956085,
756
+ "learning_rate": 3.969589978748671e-06,
757
+ "loss": 6.956285095214843,
758
+ "step": 515
759
+ },
760
+ {
761
+ "epoch": 0.104,
762
+ "grad_norm": 0.08181696385145187,
763
+ "learning_rate": 3.968430244549271e-06,
764
+ "loss": 6.970890808105469,
765
+ "step": 520
766
+ },
767
+ {
768
+ "epoch": 0.105,
769
+ "grad_norm": 0.06938958913087845,
770
+ "learning_rate": 3.967248983912086e-06,
771
+ "loss": 6.917367553710937,
772
+ "step": 525
773
+ },
774
+ {
775
+ "epoch": 0.106,
776
+ "grad_norm": 0.10486883670091629,
777
+ "learning_rate": 3.966046209755195e-06,
778
+ "loss": 6.8495323181152346,
779
+ "step": 530
780
+ },
781
+ {
782
+ "epoch": 0.107,
783
+ "grad_norm": 0.08114153891801834,
784
+ "learning_rate": 3.964821935231942e-06,
785
+ "loss": 7.0064338684082035,
786
+ "step": 535
787
+ },
788
+ {
789
+ "epoch": 0.108,
790
+ "grad_norm": 0.07422944903373718,
791
+ "learning_rate": 3.963576173730798e-06,
792
+ "loss": 6.89477767944336,
793
+ "step": 540
794
+ },
795
+ {
796
+ "epoch": 0.109,
797
+ "grad_norm": 0.0750703290104866,
798
+ "learning_rate": 3.9623089388752105e-06,
799
+ "loss": 7.047392272949219,
800
+ "step": 545
801
+ },
802
+ {
803
+ "epoch": 0.11,
804
+ "grad_norm": 0.07453129440546036,
805
+ "learning_rate": 3.961020244523458e-06,
806
+ "loss": 6.892961883544922,
807
+ "step": 550
808
+ },
809
+ {
810
+ "epoch": 0.111,
811
+ "grad_norm": 0.08709228038787842,
812
+ "learning_rate": 3.959710104768493e-06,
813
+ "loss": 6.973309326171875,
814
+ "step": 555
815
+ },
816
+ {
817
+ "epoch": 0.112,
818
+ "grad_norm": 0.06008852645754814,
819
+ "learning_rate": 3.958378533937797e-06,
820
+ "loss": 6.845113372802734,
821
+ "step": 560
822
+ },
823
+ {
824
+ "epoch": 0.113,
825
+ "grad_norm": 0.09065559506416321,
826
+ "learning_rate": 3.957025546593213e-06,
827
+ "loss": 6.88427505493164,
828
+ "step": 565
829
+ },
830
+ {
831
+ "epoch": 0.114,
832
+ "grad_norm": 0.12213204056024551,
833
+ "learning_rate": 3.9556511575307956e-06,
834
+ "loss": 6.970057678222656,
835
+ "step": 570
836
+ },
837
+ {
838
+ "epoch": 0.115,
839
+ "grad_norm": 0.10695669054985046,
840
+ "learning_rate": 3.954255381780641e-06,
841
+ "loss": 6.839192199707031,
842
+ "step": 575
843
+ },
844
+ {
845
+ "epoch": 0.116,
846
+ "grad_norm": 0.07049612700939178,
847
+ "learning_rate": 3.952838234606732e-06,
848
+ "loss": 6.857762145996094,
849
+ "step": 580
850
+ },
851
+ {
852
+ "epoch": 0.117,
853
+ "grad_norm": 0.07053842395544052,
854
+ "learning_rate": 3.951399731506761e-06,
855
+ "loss": 6.933798217773438,
856
+ "step": 585
857
+ },
858
+ {
859
+ "epoch": 0.118,
860
+ "grad_norm": 0.09034952521324158,
861
+ "learning_rate": 3.949939888211968e-06,
862
+ "loss": 6.888163757324219,
863
+ "step": 590
864
+ },
865
+ {
866
+ "epoch": 0.119,
867
+ "grad_norm": 0.10722323507070541,
868
+ "learning_rate": 3.948458720686966e-06,
869
+ "loss": 6.820646667480469,
870
+ "step": 595
871
+ },
872
+ {
873
+ "epoch": 0.12,
874
+ "grad_norm": 0.0687774196267128,
875
+ "learning_rate": 3.946956245129566e-06,
876
+ "loss": 6.872308349609375,
877
+ "step": 600
878
+ },
879
+ {
880
+ "epoch": 0.12,
881
+ "eval_accuracy": 0.13161660561660563,
882
+ "eval_loss": 6.804003715515137,
883
+ "eval_runtime": 10.4302,
884
+ "eval_samples_per_second": 9.588,
885
+ "eval_steps_per_second": 1.63,
886
+ "step": 600
887
  }
888
  ],
889
  "logging_steps": 5,