CodeIsAbstract commited on
Commit
be920f7
·
verified ·
1 Parent(s): 57984df

Training in progress, step 12000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:903447f03d58f6091e77f6c8e86618532694f7851f4bae993e86c4ce18c6f156
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c622298879cc833d1bdcc0a29bd5f4d753b0f9b392bcfe41b3a3720e37227629
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:43d16ca2f5f37c1c8859b2af1e7c37a0df1c1e053e9beab3ae0485a01ee4349e
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f770627df18e9419658687702329dfc460bbc5339dfe15ca4d75788aa1849e5a
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:78c70ce0d856bbf974b0d8a026254698e4f48063d82d2b6e496aefa99327d481
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e52021f1e1b36f10ce9c865a52c47d6a63c888218cae64f20627598071e5030f
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4bd014c1a4ed62bfbd75417f39dbd6d002ac8b7b9a0b143e7b7bcaa971ce4138
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:382f69ae5e5420217ac9e6a070a414359d614c97fe4b4d34a11abdd073f1e7c2
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2,
6
  "eval_steps": 1000,
7
- "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -798,6 +798,164 @@
798
  "eval_samples_per_second": 342.906,
799
  "eval_steps_per_second": 21.553,
800
  "step": 10000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
801
  }
802
  ],
803
  "logging_steps": 100,
@@ -817,7 +975,7 @@
817
  "attributes": {}
818
  }
819
  },
820
- "total_flos": 3.135504384e+17,
821
  "train_batch_size": 120,
822
  "trial_name": null,
823
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.24,
6
  "eval_steps": 1000,
7
+ "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
798
  "eval_samples_per_second": 342.906,
799
  "eval_steps_per_second": 21.553,
800
  "step": 10000
801
+ },
802
+ {
803
+ "epoch": 0.202,
804
+ "grad_norm": 0.24885770678520203,
805
+ "learning_rate": 0.0005429382551153432,
806
+ "loss": 3.639742431640625,
807
+ "step": 10100
808
+ },
809
+ {
810
+ "epoch": 0.204,
811
+ "grad_norm": 0.22552932798862457,
812
+ "learning_rate": 0.0005418241804541457,
813
+ "loss": 3.6341900634765625,
814
+ "step": 10200
815
+ },
816
+ {
817
+ "epoch": 0.206,
818
+ "grad_norm": 0.24933825433254242,
819
+ "learning_rate": 0.0005407005014489372,
820
+ "loss": 3.6378778076171874,
821
+ "step": 10300
822
+ },
823
+ {
824
+ "epoch": 0.208,
825
+ "grad_norm": 0.24248844385147095,
826
+ "learning_rate": 0.0005395672627280085,
827
+ "loss": 3.630330810546875,
828
+ "step": 10400
829
+ },
830
+ {
831
+ "epoch": 0.21,
832
+ "grad_norm": 0.20998498797416687,
833
+ "learning_rate": 0.0005384245092993256,
834
+ "loss": 3.6162420654296876,
835
+ "step": 10500
836
+ },
837
+ {
838
+ "epoch": 0.212,
839
+ "grad_norm": 0.23281893134117126,
840
+ "learning_rate": 0.0005372722865487427,
841
+ "loss": 3.6330096435546877,
842
+ "step": 10600
843
+ },
844
+ {
845
+ "epoch": 0.214,
846
+ "grad_norm": 0.2150452584028244,
847
+ "learning_rate": 0.0005361106402382001,
848
+ "loss": 3.625997314453125,
849
+ "step": 10700
850
+ },
851
+ {
852
+ "epoch": 0.216,
853
+ "grad_norm": 0.2066139578819275,
854
+ "learning_rate": 0.0005349396165039064,
855
+ "loss": 3.594141540527344,
856
+ "step": 10800
857
+ },
858
+ {
859
+ "epoch": 0.218,
860
+ "grad_norm": 0.23046144843101501,
861
+ "learning_rate": 0.0005337592618545055,
862
+ "loss": 3.598072204589844,
863
+ "step": 10900
864
+ },
865
+ {
866
+ "epoch": 0.22,
867
+ "grad_norm": 0.2208685725927353,
868
+ "learning_rate": 0.0005325696231692306,
869
+ "loss": 3.6059771728515626,
870
+ "step": 11000
871
+ },
872
+ {
873
+ "epoch": 0.22,
874
+ "eval_accuracy": 0.3537567638687666,
875
+ "eval_loss": 3.57378888130188,
876
+ "eval_runtime": 6.1179,
877
+ "eval_samples_per_second": 317.265,
878
+ "eval_steps_per_second": 19.941,
879
+ "step": 11000
880
+ },
881
+ {
882
+ "epoch": 0.222,
883
+ "grad_norm": 0.22270554304122925,
884
+ "learning_rate": 0.0005313707476960418,
885
+ "loss": 3.616676940917969,
886
+ "step": 11100
887
+ },
888
+ {
889
+ "epoch": 0.224,
890
+ "grad_norm": 0.20033609867095947,
891
+ "learning_rate": 0.0005301626830497491,
892
+ "loss": 3.6163037109375,
893
+ "step": 11200
894
+ },
895
+ {
896
+ "epoch": 0.226,
897
+ "grad_norm": 0.2057274878025055,
898
+ "learning_rate": 0.0005289454772101221,
899
+ "loss": 3.593800354003906,
900
+ "step": 11300
901
+ },
902
+ {
903
+ "epoch": 0.228,
904
+ "grad_norm": 0.20418649911880493,
905
+ "learning_rate": 0.000527719178519984,
906
+ "loss": 3.606725769042969,
907
+ "step": 11400
908
+ },
909
+ {
910
+ "epoch": 0.23,
911
+ "grad_norm": 0.20258976519107819,
912
+ "learning_rate": 0.000526483835683292,
913
+ "loss": 3.594194030761719,
914
+ "step": 11500
915
+ },
916
+ {
917
+ "epoch": 0.232,
918
+ "grad_norm": 0.2198181301355362,
919
+ "learning_rate": 0.0005252394977632023,
920
+ "loss": 3.596621398925781,
921
+ "step": 11600
922
+ },
923
+ {
924
+ "epoch": 0.234,
925
+ "grad_norm": 0.2138666957616806,
926
+ "learning_rate": 0.000523986214180122,
927
+ "loss": 3.5838372802734373,
928
+ "step": 11700
929
+ },
930
+ {
931
+ "epoch": 0.236,
932
+ "grad_norm": 0.22914420068264008,
933
+ "learning_rate": 0.0005227240347097463,
934
+ "loss": 3.559365539550781,
935
+ "step": 11800
936
+ },
937
+ {
938
+ "epoch": 0.238,
939
+ "grad_norm": 0.23261770606040955,
940
+ "learning_rate": 0.0005214530094810812,
941
+ "loss": 3.5759823608398436,
942
+ "step": 11900
943
+ },
944
+ {
945
+ "epoch": 0.24,
946
+ "grad_norm": 0.21132150292396545,
947
+ "learning_rate": 0.0005201731889744533,
948
+ "loss": 3.56010009765625,
949
+ "step": 12000
950
+ },
951
+ {
952
+ "epoch": 0.24,
953
+ "eval_accuracy": 0.35579436830733646,
954
+ "eval_loss": 3.5506765842437744,
955
+ "eval_runtime": 5.6754,
956
+ "eval_samples_per_second": 342.004,
957
+ "eval_steps_per_second": 21.496,
958
+ "step": 12000
959
  }
960
  ],
961
  "logging_steps": 100,
 
975
  "attributes": {}
976
  }
977
  },
978
+ "total_flos": 3.7626052608e+17,
979
  "train_batch_size": 120,
980
  "trial_name": null,
981
  "trial_params": null