CodeIsAbstract commited on
Commit
91e48ee
·
verified ·
1 Parent(s): 46a03d9

Training in progress, step 12000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1010df5598a57f3107191c355d0821d0298565ed53102d05d42a80266f4acd01
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e4ba67ca4bf7c6745ce3da9bc2e2f0a0c83e78b6b644d910324362c803bcf9f2
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:66d9db11aeb77fe2fa3bbb095fdaa33941577676015deecc0a95e5769fbe593a
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:befa4bc7ac75a8200a418f6ce90c047aad5ae982fbd5659cc7c120634c58eaec
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:61466e9412c0d194297480cf7f0e3a714d962a2892349d2aadf71260158ae2fe
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e98febf4471a4dec1877ac02b945fce7c4deb248cf77bee1819b50cbf8a3806
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a7eb88890e8a1c0d8205803f09e7feb7113f18f444391d019b68607e3c736e37
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bf0257cbbbe26fb21736af967573c2602d7e7f0f3c9ac904351cb86bd58064a3
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cb18b777ebcd388d89ae9a04b334b4e4327bc4436d4f89f04f67b009aa0dea99
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc677d4cbf29c6599dc35f2fc4173c838893b9ae384275df973375a34da97e73
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.6666666666666666,
6
  "eval_steps": 1000,
7
- "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -798,6 +798,164 @@
798
  "eval_samples_per_second": 7.614,
799
  "eval_steps_per_second": 0.381,
800
  "step": 10000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
801
  }
802
  ],
803
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.8,
6
  "eval_steps": 1000,
7
+ "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
798
  "eval_samples_per_second": 7.614,
799
  "eval_steps_per_second": 0.381,
800
  "step": 10000
801
+ },
802
+ {
803
+ "epoch": 0.6733333333333333,
804
+ "grad_norm": 0.9764471650123596,
805
+ "learning_rate": 0.00010582031340402582,
806
+ "loss": 8.11692138671875,
807
+ "step": 10100
808
+ },
809
+ {
810
+ "epoch": 0.68,
811
+ "grad_norm": 1.0789597034454346,
812
+ "learning_rate": 0.00010195372756105512,
813
+ "loss": 8.137109375,
814
+ "step": 10200
815
+ },
816
+ {
817
+ "epoch": 0.6866666666666666,
818
+ "grad_norm": 1.0786553621292114,
819
+ "learning_rate": 9.813479397926537e-05,
820
+ "loss": 8.17680419921875,
821
+ "step": 10300
822
+ },
823
+ {
824
+ "epoch": 0.6933333333333334,
825
+ "grad_norm": 1.3332428932189941,
826
+ "learning_rate": 9.436536872942766e-05,
827
+ "loss": 8.15776123046875,
828
+ "step": 10400
829
+ },
830
+ {
831
+ "epoch": 0.7,
832
+ "grad_norm": 1.0370911359786987,
833
+ "learning_rate": 9.064728382036833e-05,
834
+ "loss": 8.137608642578124,
835
+ "step": 10500
836
+ },
837
+ {
838
+ "epoch": 0.7066666666666667,
839
+ "grad_norm": 1.081002950668335,
840
+ "learning_rate": 8.698234630857997e-05,
841
+ "loss": 8.0797705078125,
842
+ "step": 10600
843
+ },
844
+ {
845
+ "epoch": 0.7133333333333334,
846
+ "grad_norm": 1.0866549015045166,
847
+ "learning_rate": 8.337233741995907e-05,
848
+ "loss": 8.10765625,
849
+ "step": 10700
850
+ },
851
+ {
852
+ "epoch": 0.72,
853
+ "grad_norm": 1.4178708791732788,
854
+ "learning_rate": 7.9819011684098e-05,
855
+ "loss": 8.233140258789062,
856
+ "step": 10800
857
+ },
858
+ {
859
+ "epoch": 0.7266666666666667,
860
+ "grad_norm": 1.0158106088638306,
861
+ "learning_rate": 7.632409608155228e-05,
862
+ "loss": 8.12737060546875,
863
+ "step": 10900
864
+ },
865
+ {
866
+ "epoch": 0.7333333333333333,
867
+ "grad_norm": 1.0095257759094238,
868
+ "learning_rate": 7.288928920449636e-05,
869
+ "loss": 8.0778955078125,
870
+ "step": 11000
871
+ },
872
+ {
873
+ "epoch": 0.7333333333333333,
874
+ "eval_accuracy": 0.3137651663405088,
875
+ "eval_loss": 8.122584342956543,
876
+ "eval_runtime": 65.3252,
877
+ "eval_samples_per_second": 7.654,
878
+ "eval_steps_per_second": 0.383,
879
+ "step": 11000
880
+ },
881
+ {
882
+ "epoch": 0.74,
883
+ "grad_norm": 1.1707996129989624,
884
+ "learning_rate": 6.951626043117705e-05,
885
+ "loss": 8.111912841796874,
886
+ "step": 11100
887
+ },
888
+ {
889
+ "epoch": 0.7466666666666667,
890
+ "grad_norm": 1.1220096349716187,
891
+ "learning_rate": 6.620664911456616e-05,
892
+ "loss": 8.153399047851563,
893
+ "step": 11200
894
+ },
895
+ {
896
+ "epoch": 0.7533333333333333,
897
+ "grad_norm": 1.053057312965393,
898
+ "learning_rate": 6.296206378560454e-05,
899
+ "loss": 8.083673095703125,
900
+ "step": 11300
901
+ },
902
+ {
903
+ "epoch": 0.76,
904
+ "grad_norm": 1.0960450172424316,
905
+ "learning_rate": 5.978408137142759e-05,
906
+ "loss": 8.193185424804687,
907
+ "step": 11400
908
+ },
909
+ {
910
+ "epoch": 0.7666666666666667,
911
+ "grad_norm": 1.1686227321624756,
912
+ "learning_rate": 5.667424642894974e-05,
913
+ "loss": 8.13611572265625,
914
+ "step": 11500
915
+ },
916
+ {
917
+ "epoch": 0.7733333333333333,
918
+ "grad_norm": 1.103230595588684,
919
+ "learning_rate": 5.363407039418178e-05,
920
+ "loss": 8.102085571289063,
921
+ "step": 11600
922
+ },
923
+ {
924
+ "epoch": 0.78,
925
+ "grad_norm": 1.3868330717086792,
926
+ "learning_rate": 5.0665030847646266e-05,
927
+ "loss": 8.002344970703126,
928
+ "step": 11700
929
+ },
930
+ {
931
+ "epoch": 0.7866666666666666,
932
+ "grad_norm": 1.0746208429336548,
933
+ "learning_rate": 4.776857079624599e-05,
934
+ "loss": 8.036705322265625,
935
+ "step": 11800
936
+ },
937
+ {
938
+ "epoch": 0.7933333333333333,
939
+ "grad_norm": 1.161407470703125,
940
+ "learning_rate": 4.494609797193681e-05,
941
+ "loss": 7.990054931640625,
942
+ "step": 11900
943
+ },
944
+ {
945
+ "epoch": 0.8,
946
+ "grad_norm": 1.1083136796951294,
947
+ "learning_rate": 4.219898414754464e-05,
948
+ "loss": 8.060489501953125,
949
+ "step": 12000
950
+ },
951
+ {
952
+ "epoch": 0.8,
953
+ "eval_accuracy": 0.3169706457925636,
954
+ "eval_loss": 8.059541702270508,
955
+ "eval_runtime": 65.8107,
956
+ "eval_samples_per_second": 7.598,
957
+ "eval_steps_per_second": 0.38,
958
+ "step": 12000
959
  }
960
  ],
961
  "logging_steps": 100,