CodeIsAbstract commited on
Commit
8b7a709
·
verified ·
1 Parent(s): eceb1b4

Training in progress, step 10000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bc0c1563393d995412311cba1214a9520dca547b90b519de7f1ce0b2f94384a6
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8dc9b7da6d26d633c2dd1a6298a08b45f5c5918ae1aaa6d997b662eb648987d
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3aa9b056f2e17085d4900f3271d88de35f6ac5194621790395f2e4f587a69232
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7e5109c5ab50d16612c916165b2f1b379e87b9655586367244ed0f35eed2e9a8
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1e47a193d21ecc1ad99cf3d0b11067079c7e2eeed011077dd9d2eec110457c9c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2650365deef6e7f360be9092deb581b50fcfc0981b456b54de45695f1dd87947
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:768ce36046f98e2d652c3d4e4715ee92969f1d46b33288d5213930731a801cf9
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d9300f32e301bd335f3fb5d2ba765ba95180c24ea46eaea685e58401c90015c1
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ad69461fda5a7151f961131a57f67c28a9bf598350ae5d99ce95c8ec88ff78da
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f730639438ee8fcdcc1e1cb83751cf1eada38bbe0ef956dce2851ca931960b8a
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4f092d919dcd7a979986c37fe2f4d5718eda679a575e6e71176605e3f28b4da3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fe934d44769ac7e9a840b774edada5ec3bca8cf109f459667d3c8dc00da8820d
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7cfd234ef8e0bce0d28c2272f50bbffc1faa98bd25482e7a5eb6e1a7d18e6aab
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b250b2b78856a4178b9411fc4c2c48db7ef90f84dc2088843965ef7dfde0aa6
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9277cfd5b2385bc7162ccde048b9e10c1f018201304953d47b48ffdd2baae832
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67ca67d6482b0321beff26f4299075e123e65721ad7171ad8e9b4413b619f027
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:110f59eb6a8f1b021131a112cdfbbed170d6a8d222fbcb0a48b72e3683fe424f
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f545309b770f4b9aabc99027a1da637fdad2e36ae6df1f09d6f3e87c09353442
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:28db19b12e448a6216bc82ad0a8dc000f968411ed4a0fc53a441e099e33a6ca8
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b930024c4c4916aafefd00a76494fc84038df46cd42e5e9dc314f9cb1f7dfe3
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4dc3ee7ff07df340dc973522b19f5110e955054f7179c9fdb7ad1c0dee99bc27
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82c259140dc729f7726e15f02d46430c5619fa08380b6000b04b9de5c5a519c4
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.32,
6
  "eval_steps": 1000,
7
- "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -640,6 +640,164 @@
640
  "eval_samples_per_second": 17.907,
641
  "eval_steps_per_second": 0.573,
642
  "step": 8000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
643
  }
644
  ],
645
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4,
6
  "eval_steps": 1000,
7
+ "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
640
  "eval_samples_per_second": 17.907,
641
  "eval_steps_per_second": 0.573,
642
  "step": 8000
643
+ },
644
+ {
645
+ "epoch": 0.324,
646
+ "grad_norm": 3.4113783836364746,
647
+ "learning_rate": 0.0030520091191551142,
648
+ "loss": 45.00322265625,
649
+ "step": 8100
650
+ },
651
+ {
652
+ "epoch": 0.328,
653
+ "grad_norm": 3.566800832748413,
654
+ "learning_rate": 0.003030543062602622,
655
+ "loss": 45.1212841796875,
656
+ "step": 8200
657
+ },
658
+ {
659
+ "epoch": 0.332,
660
+ "grad_norm": 3.9237873554229736,
661
+ "learning_rate": 0.003008914141089914,
662
+ "loss": 45.1877978515625,
663
+ "step": 8300
664
+ },
665
+ {
666
+ "epoch": 0.336,
667
+ "grad_norm": 4.103876113891602,
668
+ "learning_rate": 0.0029871257728083995,
669
+ "loss": 45.1942626953125,
670
+ "step": 8400
671
+ },
672
+ {
673
+ "epoch": 0.34,
674
+ "grad_norm": 3.259410858154297,
675
+ "learning_rate": 0.0029651814011481333,
676
+ "loss": 44.929365234375,
677
+ "step": 8500
678
+ },
679
+ {
680
+ "epoch": 0.344,
681
+ "grad_norm": 3.518616199493408,
682
+ "learning_rate": 0.002943084494153631,
683
+ "loss": 44.5111328125,
684
+ "step": 8600
685
+ },
686
+ {
687
+ "epoch": 0.348,
688
+ "grad_norm": 3.544551372528076,
689
+ "learning_rate": 0.002920838543975789,
690
+ "loss": 44.8691015625,
691
+ "step": 8700
692
+ },
693
+ {
694
+ "epoch": 0.352,
695
+ "grad_norm": 3.5976264476776123,
696
+ "learning_rate": 0.0028984470663199887,
697
+ "loss": 44.924892578125,
698
+ "step": 8800
699
+ },
700
+ {
701
+ "epoch": 0.356,
702
+ "grad_norm": 3.3831348419189453,
703
+ "learning_rate": 0.0028759135998904796,
704
+ "loss": 44.773828125,
705
+ "step": 8900
706
+ },
707
+ {
708
+ "epoch": 0.36,
709
+ "grad_norm": 3.3760013580322266,
710
+ "learning_rate": 0.002853241705831138,
711
+ "loss": 44.8477294921875,
712
+ "step": 9000
713
+ },
714
+ {
715
+ "epoch": 0.36,
716
+ "eval_accuracy": 0.2051945259042033,
717
+ "eval_loss": 44.57013702392578,
718
+ "eval_runtime": 7.2683,
719
+ "eval_samples_per_second": 17.198,
720
+ "eval_steps_per_second": 0.55,
721
+ "step": 9000
722
+ },
723
+ {
724
+ "epoch": 0.364,
725
+ "grad_norm": 2.7893764972686768,
726
+ "learning_rate": 0.0028304349671626613,
727
+ "loss": 44.7609765625,
728
+ "step": 9100
729
+ },
730
+ {
731
+ "epoch": 0.368,
732
+ "grad_norm": 3.8668603897094727,
733
+ "learning_rate": 0.002807496988216319,
734
+ "loss": 44.6481201171875,
735
+ "step": 9200
736
+ },
737
+ {
738
+ "epoch": 0.372,
739
+ "grad_norm": 3.4443302154541016,
740
+ "learning_rate": 0.002784431394064333,
741
+ "loss": 44.4382373046875,
742
+ "step": 9300
743
+ },
744
+ {
745
+ "epoch": 0.376,
746
+ "grad_norm": 3.2611913681030273,
747
+ "learning_rate": 0.0027612418299469746,
748
+ "loss": 44.3794384765625,
749
+ "step": 9400
750
+ },
751
+ {
752
+ "epoch": 0.38,
753
+ "grad_norm": 2.5774736404418945,
754
+ "learning_rate": 0.0027379319606964806,
755
+ "loss": 44.4774462890625,
756
+ "step": 9500
757
+ },
758
+ {
759
+ "epoch": 0.384,
760
+ "grad_norm": 3.2907772064208984,
761
+ "learning_rate": 0.002714505470157871,
762
+ "loss": 44.4849267578125,
763
+ "step": 9600
764
+ },
765
+ {
766
+ "epoch": 0.388,
767
+ "grad_norm": 3.4475579261779785,
768
+ "learning_rate": 0.0026909660606067583,
769
+ "loss": 44.4125537109375,
770
+ "step": 9700
771
+ },
772
+ {
773
+ "epoch": 0.392,
774
+ "grad_norm": 4.186856269836426,
775
+ "learning_rate": 0.002667317452164251,
776
+ "loss": 44.404677734375,
777
+ "step": 9800
778
+ },
779
+ {
780
+ "epoch": 0.396,
781
+ "grad_norm": 3.1886963844299316,
782
+ "learning_rate": 0.0026435633822090316,
783
+ "loss": 44.2500146484375,
784
+ "step": 9900
785
+ },
786
+ {
787
+ "epoch": 0.4,
788
+ "grad_norm": 3.3009133338928223,
789
+ "learning_rate": 0.002619707604786708,
790
+ "loss": 44.2538232421875,
791
+ "step": 10000
792
+ },
793
+ {
794
+ "epoch": 0.4,
795
+ "eval_accuracy": 0.20702737047898337,
796
+ "eval_loss": 44.112098693847656,
797
+ "eval_runtime": 7.3398,
798
+ "eval_samples_per_second": 17.03,
799
+ "eval_steps_per_second": 0.545,
800
+ "step": 10000
801
  }
802
  ],
803
  "logging_steps": 100,