CodeIsAbstract commited on
Commit
a89da1c
·
verified ·
1 Parent(s): 757dafb

Training in progress, step 12000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ff5f848615af7284182bf34bb526cf66a4611210963f90a45f3c9021871e1d85
3
  size 1600779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:895b25ff701dcc070741cf42351fb18c429fa56d815b15c7ba3c867f23a10980
3
  size 1600779
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2b4c790553f3e4092f7857a64225265bb38246a5d8a25fe76b1107d5f0509e1f
3
  size 621149155
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1313ea19ff807c10147842a59b9051e7021a8c028fe68e36adab013243f36b38
3
  size 621149155
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:189ca5f500a151c520561b912184d219fb4a14e37e0545ece3a7916f40a7b6f6
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f794b047026bd2e455c2f0c1a40191bfadaa2e43fd08fedee4185a5059db1b48
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:585e6d762d92f384f163db69802c96c5f70ee11af325886531da1e25dbdd3c46
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a5541088e5d274181d6c9ef46a5b103eb5b8bf402d2abae7634635f86f3865ab
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:23411dc088849538602b2c97c02e11f26fb822081e4a284d2b43ea9891ec5924
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da6b2d1e9271342e1e302bb1b26316b143cfc6c4934d2ed389c10bb161aa25ef
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a96d850263b64569ae2096b8d72bc91a5ebdd71deb02d3267c85822dd6891262
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7900dd4bec725410e9ff4f8c55274ea9809b523a68297edbd3a83622136595cd
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:14acad3fcc328519114bc488b787139326de9b018d7a082a9a787fbad2c61513
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d05d56cf9a1650f01c1af9b1a787ec6db7126c64049f267e937381c9944e1282
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4d9d4bb8960e987a9f4826d17f0ffbf80325efd81a2b4fbccc2aba1b74b8cb51
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9be3515ba8ff34af4f0c40faacd8a5a9e8e607850863dcb77d772f0606a1ae0c
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ca2326a5fbe6bf3c05da0cff924b9b40c864d3c0c3f624c13f7edec4700ef239
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2e3e189e1259797127a8750045cf99ca33e22e9a0c01f4e0afcdf1fa22431dd
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:df1ed186896b3430710bec5fd2c1b85c275cbb3ef74ad87064fdc0302116db5c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a9afcf27fe2dffd21d9ec60d55b8d2840fcdcf2a30cdfc9362419caa2813c93
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:833776bf99a0597a4bf38df86be38e8ac151736eba1d084c9cbb84cab1caf4e9
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:43ab2089bb2cb7b873a77ac8e24a3b3b2416ededca3811b6bb0922f61bea4c99
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2,
6
  "eval_steps": 1000,
7
- "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -798,6 +798,164 @@
798
  "eval_samples_per_second": 39.294,
799
  "eval_steps_per_second": 0.629,
800
  "step": 10000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
801
  }
802
  ],
803
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.24,
6
  "eval_steps": 1000,
7
+ "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
798
  "eval_samples_per_second": 39.294,
799
  "eval_steps_per_second": 0.629,
800
  "step": 10000
801
+ },
802
+ {
803
+ "epoch": 0.202,
804
+ "grad_norm": 1.518142819404602,
805
+ "learning_rate": 0.0018763385407335963,
806
+ "loss": 32.7453,
807
+ "step": 10100
808
+ },
809
+ {
810
+ "epoch": 0.204,
811
+ "grad_norm": 1.908083200454712,
812
+ "learning_rate": 0.001873133519711602,
813
+ "loss": 32.9035,
814
+ "step": 10200
815
+ },
816
+ {
817
+ "epoch": 0.206,
818
+ "grad_norm": 2.4479877948760986,
819
+ "learning_rate": 0.0018698903050008956,
820
+ "loss": 32.3298,
821
+ "step": 10300
822
+ },
823
+ {
824
+ "epoch": 0.208,
825
+ "grad_norm": 1.403601050376892,
826
+ "learning_rate": 0.0018666090384701947,
827
+ "loss": 32.0576,
828
+ "step": 10400
829
+ },
830
+ {
831
+ "epoch": 0.21,
832
+ "grad_norm": 1.7655203342437744,
833
+ "learning_rate": 0.001863289863652727,
834
+ "loss": 32.8307,
835
+ "step": 10500
836
+ },
837
+ {
838
+ "epoch": 0.212,
839
+ "grad_norm": 1.317898154258728,
840
+ "learning_rate": 0.0018599329257399516,
841
+ "loss": 32.8206,
842
+ "step": 10600
843
+ },
844
+ {
845
+ "epoch": 0.214,
846
+ "grad_norm": 1.6275999546051025,
847
+ "learning_rate": 0.0018565383715752083,
848
+ "loss": 33.1208,
849
+ "step": 10700
850
+ },
851
+ {
852
+ "epoch": 0.216,
853
+ "grad_norm": 1.350576639175415,
854
+ "learning_rate": 0.0018531063496472927,
855
+ "loss": 33.09,
856
+ "step": 10800
857
+ },
858
+ {
859
+ "epoch": 0.218,
860
+ "grad_norm": 2.071432590484619,
861
+ "learning_rate": 0.0018496370100839622,
862
+ "loss": 32.4823,
863
+ "step": 10900
864
+ },
865
+ {
866
+ "epoch": 0.22,
867
+ "grad_norm": 1.756463646888733,
868
+ "learning_rate": 0.0018461305046453683,
869
+ "loss": 32.3173,
870
+ "step": 11000
871
+ },
872
+ {
873
+ "epoch": 0.22,
874
+ "eval_accuracy": 0.29537964774951075,
875
+ "eval_loss": 32.61259078979492,
876
+ "eval_runtime": 3.1999,
877
+ "eval_samples_per_second": 39.064,
878
+ "eval_steps_per_second": 0.625,
879
+ "step": 11000
880
+ },
881
+ {
882
+ "epoch": 0.222,
883
+ "grad_norm": 2.6366186141967773,
884
+ "learning_rate": 0.0018425869867174187,
885
+ "loss": 32.0905,
886
+ "step": 11100
887
+ },
888
+ {
889
+ "epoch": 0.224,
890
+ "grad_norm": 1.3121482133865356,
891
+ "learning_rate": 0.0018390066113050665,
892
+ "loss": 32.2553,
893
+ "step": 11200
894
+ },
895
+ {
896
+ "epoch": 0.226,
897
+ "grad_norm": 8.878124237060547,
898
+ "learning_rate": 0.0018353895350255317,
899
+ "loss": 33.1959,
900
+ "step": 11300
901
+ },
902
+ {
903
+ "epoch": 0.228,
904
+ "grad_norm": 2.0059814453125,
905
+ "learning_rate": 0.0018317359161014477,
906
+ "loss": 32.804,
907
+ "step": 11400
908
+ },
909
+ {
910
+ "epoch": 0.23,
911
+ "grad_norm": 1.3689004182815552,
912
+ "learning_rate": 0.001828045914353943,
913
+ "loss": 32.6974,
914
+ "step": 11500
915
+ },
916
+ {
917
+ "epoch": 0.232,
918
+ "grad_norm": 1.6248316764831543,
919
+ "learning_rate": 0.0018243196911956476,
920
+ "loss": 32.5622,
921
+ "step": 11600
922
+ },
923
+ {
924
+ "epoch": 0.234,
925
+ "grad_norm": 1.3218642473220825,
926
+ "learning_rate": 0.0018205574096236336,
927
+ "loss": 32.348,
928
+ "step": 11700
929
+ },
930
+ {
931
+ "epoch": 0.236,
932
+ "grad_norm": 1.4009215831756592,
933
+ "learning_rate": 0.0018167592342122857,
934
+ "loss": 31.677,
935
+ "step": 11800
936
+ },
937
+ {
938
+ "epoch": 0.238,
939
+ "grad_norm": 1.8492565155029297,
940
+ "learning_rate": 0.0018129253311061002,
941
+ "loss": 31.6381,
942
+ "step": 11900
943
+ },
944
+ {
945
+ "epoch": 0.24,
946
+ "grad_norm": 2.3902571201324463,
947
+ "learning_rate": 0.0018090558680124193,
948
+ "loss": 32.4851,
949
+ "step": 12000
950
+ },
951
+ {
952
+ "epoch": 0.24,
953
+ "eval_accuracy": 0.300720156555773,
954
+ "eval_loss": 32.26734161376953,
955
+ "eval_runtime": 3.3297,
956
+ "eval_samples_per_second": 37.54,
957
+ "eval_steps_per_second": 0.601,
958
+ "step": 12000
959
  }
960
  ],
961
  "logging_steps": 100,