Professor commited on
Commit
48cdfa1
·
verified ·
1 Parent(s): 1675876

Training in progress, step 60000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:68a8797cd76a94ad3c6a56a7adc2bc1ac7c0398bdd3f9b8fd64c36634fe35a09
3
  size 242247040
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:37c3e862658b65ca1c890f9f0b04bb5662ecbde856aff6689ee9869db46d9afb
3
  size 242247040
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:97396477b00599023e8b9271f179e4f9045180aadb813660ee60fd378f08ad79
3
  size 484395467
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bb4f4c4c584d8852a4f54227a4581f8060bac83c9d018fe6714f1e4cfab0c599
3
  size 484395467
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ea14e437e9bde41dd138cf43e193fe8b9138dbdf51219ee5b3285011252c1d78
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:45f40ff98c14fbbe6809e0534caa0e9d45b0cb19d01a65ef2e74b4fd68c0c624
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4f695b69ce20f0a3ab929251148b7f644620cd816442ed26ba4e618a194a3b03
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:72db15717b2d4617f35951f97b2fabd7d047d3b7fe2b49b17504acc7cc4ec68d
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:51d2908fdfd796e12d1265cbd5e8f7336ad5dbe4ebdb91a07beab91214e64999
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b62892e5bace3013f6f9fb7a156e9e583088be1e4d87080b99784c2eb93c454b
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
- "best_global_step": 55000,
3
- "best_metric": 25.42,
4
- "best_model_checkpoint": "/content/en_kin_model/checkpoint-55000",
5
- "epoch": 12.999290947766486,
6
  "eval_steps": 5000,
7
- "global_step": 55000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -877,6 +877,85 @@
877
  "eval_samples_per_second": 138.762,
878
  "eval_steps_per_second": 1.105,
879
  "step": 55000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
880
  }
881
  ],
882
  "logging_steps": 500,
@@ -905,7 +984,7 @@
905
  "attributes": {}
906
  }
907
  },
908
- "total_flos": 1.9997212897876378e+17,
909
  "train_batch_size": 128,
910
  "trial_name": null,
911
  "trial_params": null
 
1
  {
2
+ "best_global_step": 60000,
3
+ "best_metric": 25.69,
4
+ "best_model_checkpoint": "/content/en_kin_model/checkpoint-60000",
5
+ "epoch": 14.181044670290712,
6
  "eval_steps": 5000,
7
+ "global_step": 60000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
877
  "eval_samples_per_second": 138.762,
878
  "eval_steps_per_second": 1.105,
879
  "step": 55000
880
+ },
881
+ {
882
+ "epoch": 13.117466320018908,
883
+ "grad_norm": 0.72737056016922,
884
+ "learning_rate": 0.0003615810122196134,
885
+ "loss": 2.507847412109375,
886
+ "step": 55500
887
+ },
888
+ {
889
+ "epoch": 13.235641692271331,
890
+ "grad_norm": 0.7681657075881958,
891
+ "learning_rate": 0.00035916764747355705,
892
+ "loss": 2.53956298828125,
893
+ "step": 56000
894
+ },
895
+ {
896
+ "epoch": 13.353817064523753,
897
+ "grad_norm": 0.8309329748153687,
898
+ "learning_rate": 0.0003567416463147021,
899
+ "loss": 2.55163134765625,
900
+ "step": 56500
901
+ },
902
+ {
903
+ "epoch": 13.471992436776176,
904
+ "grad_norm": 0.7395633459091187,
905
+ "learning_rate": 0.0003543032895584072,
906
+ "loss": 2.56509521484375,
907
+ "step": 57000
908
+ },
909
+ {
910
+ "epoch": 13.590167809028598,
911
+ "grad_norm": 0.7688890099525452,
912
+ "learning_rate": 0.0003518528594502208,
913
+ "loss": 2.564052734375,
914
+ "step": 57500
915
+ },
916
+ {
917
+ "epoch": 13.708343181281021,
918
+ "grad_norm": 0.7620258331298828,
919
+ "learning_rate": 0.00034939063963321005,
920
+ "loss": 2.583301513671875,
921
+ "step": 58000
922
+ },
923
+ {
924
+ "epoch": 13.826518553533443,
925
+ "grad_norm": 0.7235142588615417,
926
+ "learning_rate": 0.0003469169151151291,
927
+ "loss": 2.5848076171875,
928
+ "step": 58500
929
+ },
930
+ {
931
+ "epoch": 13.944693925785867,
932
+ "grad_norm": 0.7277477979660034,
933
+ "learning_rate": 0.00034443197223542793,
934
+ "loss": 2.59660107421875,
935
+ "step": 59000
936
+ },
937
+ {
938
+ "epoch": 14.062869298038288,
939
+ "grad_norm": 0.6620763540267944,
940
+ "learning_rate": 0.0003419360986321088,
941
+ "loss": 2.519939697265625,
942
+ "step": 59500
943
+ },
944
+ {
945
+ "epoch": 14.181044670290712,
946
+ "grad_norm": 0.7132055163383484,
947
+ "learning_rate": 0.0003394295832084308,
948
+ "loss": 2.478825439453125,
949
+ "step": 60000
950
+ },
951
+ {
952
+ "epoch": 14.181044670290712,
953
+ "eval_bleu": 25.69,
954
+ "eval_loss": 2.8314549922943115,
955
+ "eval_runtime": 39.5905,
956
+ "eval_samples_per_second": 139.554,
957
+ "eval_steps_per_second": 1.111,
958
+ "step": 60000
959
  }
960
  ],
961
  "logging_steps": 500,
 
984
  "attributes": {}
985
  }
986
  },
987
+ "total_flos": 2.1813850903648666e+17,
988
  "train_batch_size": 128,
989
  "trial_name": null,
990
  "trial_params": null