CodeIsAbstract commited on
Commit
88e253a
·
verified ·
1 Parent(s): b573e92

Training in progress, step 12000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c8dc9b7da6d26d633c2dd1a6298a08b45f5c5918ae1aaa6d997b662eb648987d
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b8e8d3ec310b587a737b2ff1479785c95b383b0103a20c2616bcaa1b70cedfd
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e5109c5ab50d16612c916165b2f1b379e87b9655586367244ed0f35eed2e9a8
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af1c4c4789099e3f06b7040705745688787a84de28b3c92b707a08bf66acedd5
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2650365deef6e7f360be9092deb581b50fcfc0981b456b54de45695f1dd87947
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de97c976c1994a7f2b52cd661dd45e50f8c59ea7f29046e2b153d031a5329003
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d9300f32e301bd335f3fb5d2ba765ba95180c24ea46eaea685e58401c90015c1
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:945dfe762c470357f0955a0e4fddcfadf68d636dfea2666def3fd34df1ab5189
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f730639438ee8fcdcc1e1cb83751cf1eada38bbe0ef956dce2851ca931960b8a
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3883b9681a4e564010f0c080b9a5c164f6cfc69d29380b6cdfc7bd14e3061a6c
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fe934d44769ac7e9a840b774edada5ec3bca8cf109f459667d3c8dc00da8820d
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1bfa4cd8ff4ea106b4895cc38559e225da4381a514e25136b0ad798f67a76ec0
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2b250b2b78856a4178b9411fc4c2c48db7ef90f84dc2088843965ef7dfde0aa6
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7bd3539150b8b4416c1ef3e8435df0b2b18db72a3b53e80a325b1d948246653b
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:67ca67d6482b0321beff26f4299075e123e65721ad7171ad8e9b4413b619f027
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:414056302bb2c51030907ea288f267bbdafb5b349fed9bbe5df32046e41df462
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f545309b770f4b9aabc99027a1da637fdad2e36ae6df1f09d6f3e87c09353442
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:217d507942f28f310e4eeed6bfba8d07fd2bb6dc167c4c4f66eda1c08b6318d6
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7b930024c4c4916aafefd00a76494fc84038df46cd42e5e9dc314f9cb1f7dfe3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b43a58b72baf558b7e5624a89206c612752404e0e3755fa74aff6a8a856fea
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:82c259140dc729f7726e15f02d46430c5619fa08380b6000b04b9de5c5a519c4
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ef456277f681db2c07f1ff63be489edd84235ae448b25e8872b7c6ddb3f68afb
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4,
6
  "eval_steps": 1000,
7
- "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -798,6 +798,164 @@
798
  "eval_samples_per_second": 17.03,
799
  "eval_steps_per_second": 0.545,
800
  "step": 10000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
801
  }
802
  ],
803
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.48,
6
  "eval_steps": 1000,
7
+ "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
798
  "eval_samples_per_second": 17.03,
799
  "eval_steps_per_second": 0.545,
800
  "step": 10000
801
+ },
802
+ {
803
+ "epoch": 0.404,
804
+ "grad_norm": 3.5040667057037354,
805
+ "learning_rate": 0.0025957538900165333,
806
+ "loss": 44.2118505859375,
807
+ "step": 10100
808
+ },
809
+ {
810
+ "epoch": 0.408,
811
+ "grad_norm": 3.0154404640197754,
812
+ "learning_rate": 0.0025717060234955805,
813
+ "loss": 44.339365234375,
814
+ "step": 10200
815
+ },
816
+ {
817
+ "epoch": 0.412,
818
+ "grad_norm": 3.0024242401123047,
819
+ "learning_rate": 0.0025475678057004796,
820
+ "loss": 44.2085009765625,
821
+ "step": 10300
822
+ },
823
+ {
824
+ "epoch": 0.416,
825
+ "grad_norm": 2.868852138519287,
826
+ "learning_rate": 0.002523343051386793,
827
+ "loss": 44.1264453125,
828
+ "step": 10400
829
+ },
830
+ {
831
+ "epoch": 0.42,
832
+ "grad_norm": 3.098557472229004,
833
+ "learning_rate": 0.0024990355889861417,
834
+ "loss": 44.0060986328125,
835
+ "step": 10500
836
+ },
837
+ {
838
+ "epoch": 0.424,
839
+ "grad_norm": 2.94388747215271,
840
+ "learning_rate": 0.0024746492600011666,
841
+ "loss": 44.090751953125,
842
+ "step": 10600
843
+ },
844
+ {
845
+ "epoch": 0.428,
846
+ "grad_norm": 3.172335147857666,
847
+ "learning_rate": 0.002450187918398426,
848
+ "loss": 44.108330078125,
849
+ "step": 10700
850
+ },
851
+ {
852
+ "epoch": 0.432,
853
+ "grad_norm": 3.123621702194214,
854
+ "learning_rate": 0.0024256554299993214,
855
+ "loss": 44.00998046875,
856
+ "step": 10800
857
+ },
858
+ {
859
+ "epoch": 0.436,
860
+ "grad_norm": 2.8541388511657715,
861
+ "learning_rate": 0.002401055671869152,
862
+ "loss": 43.941904296875,
863
+ "step": 10900
864
+ },
865
+ {
866
+ "epoch": 0.44,
867
+ "grad_norm": 2.895085573196411,
868
+ "learning_rate": 0.0023763925317043903,
869
+ "loss": 43.792197265625,
870
+ "step": 11000
871
+ },
872
+ {
873
+ "epoch": 0.44,
874
+ "eval_accuracy": 0.20995894428152492,
875
+ "eval_loss": 43.750606536865234,
876
+ "eval_runtime": 6.9444,
877
+ "eval_samples_per_second": 18.0,
878
+ "eval_steps_per_second": 0.576,
879
+ "step": 11000
880
+ },
881
+ {
882
+ "epoch": 0.444,
883
+ "grad_norm": 2.346236228942871,
884
+ "learning_rate": 0.0023516699072182786,
885
+ "loss": 43.9089990234375,
886
+ "step": 11100
887
+ },
888
+ {
889
+ "epoch": 0.448,
890
+ "grad_norm": 2.3568971157073975,
891
+ "learning_rate": 0.002326891705524841,
892
+ "loss": 43.76560546875,
893
+ "step": 11200
894
+ },
895
+ {
896
+ "epoch": 0.452,
897
+ "grad_norm": 3.0081124305725098,
898
+ "learning_rate": 0.0023020618425214144,
899
+ "loss": 43.808974609375,
900
+ "step": 11300
901
+ },
902
+ {
903
+ "epoch": 0.456,
904
+ "grad_norm": 3.7043468952178955,
905
+ "learning_rate": 0.0022771842422697835,
906
+ "loss": 43.849228515625,
907
+ "step": 11400
908
+ },
909
+ {
910
+ "epoch": 0.46,
911
+ "grad_norm": 2.2964367866516113,
912
+ "learning_rate": 0.0022522628363760323,
913
+ "loss": 43.599345703125,
914
+ "step": 11500
915
+ },
916
+ {
917
+ "epoch": 0.464,
918
+ "grad_norm": 2.747478485107422,
919
+ "learning_rate": 0.002227301563369202,
920
+ "loss": 43.42494140625,
921
+ "step": 11600
922
+ },
923
+ {
924
+ "epoch": 0.468,
925
+ "grad_norm": 2.878131866455078,
926
+ "learning_rate": 0.0022023043680788512,
927
+ "loss": 43.6928955078125,
928
+ "step": 11700
929
+ },
930
+ {
931
+ "epoch": 0.472,
932
+ "grad_norm": 2.552713632583618,
933
+ "learning_rate": 0.0021772752010116247,
934
+ "loss": 43.6564453125,
935
+ "step": 11800
936
+ },
937
+ {
938
+ "epoch": 0.476,
939
+ "grad_norm": 2.9993340969085693,
940
+ "learning_rate": 0.002152218017726923,
941
+ "loss": 43.607763671875,
942
+ "step": 11900
943
+ },
944
+ {
945
+ "epoch": 0.48,
946
+ "grad_norm": 2.911576747894287,
947
+ "learning_rate": 0.002127136778211772,
948
+ "loss": 43.8162890625,
949
+ "step": 12000
950
+ },
951
+ {
952
+ "epoch": 0.48,
953
+ "eval_accuracy": 0.21305474095796675,
954
+ "eval_loss": 43.4426155090332,
955
+ "eval_runtime": 7.5722,
956
+ "eval_samples_per_second": 16.508,
957
+ "eval_steps_per_second": 0.528,
958
+ "step": 12000
959
  }
960
  ],
961
  "logging_steps": 100,