CodeIsAbstract commited on
Commit
2d8a4ac
·
verified ·
1 Parent(s): 535917b

Training in progress, step 14000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4b8e8d3ec310b587a737b2ff1479785c95b383b0103a20c2616bcaa1b70cedfd
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:75ba0b7a63e25df56e29b8b1f4ae481007e80d1cbde28fce2cb4ed43b8b03d7f
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:af1c4c4789099e3f06b7040705745688787a84de28b3c92b707a08bf66acedd5
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bbeaa2e3e7f8d9416a2aff04c3f41cb4a4fbb000165b5f4b0f96556dc4f75345
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:de97c976c1994a7f2b52cd661dd45e50f8c59ea7f29046e2b153d031a5329003
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b9fd33b0410b29f1c36e49a77f8193ae40be73acb6bc07f4d04bbf2f5a48485d
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:945dfe762c470357f0955a0e4fddcfadf68d636dfea2666def3fd34df1ab5189
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3b178b887d0edcf498486b28f0a4fa9660c9ddb3e2c9439c66b45fe0ea200cf9
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3883b9681a4e564010f0c080b9a5c164f6cfc69d29380b6cdfc7bd14e3061a6c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3ffac77d94e352efd69a7a08457843666ef6e11840940a75b79aa309c60b093e
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1bfa4cd8ff4ea106b4895cc38559e225da4381a514e25136b0ad798f67a76ec0
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a4793c6e47d0cc17f59428134402f504910de29559f3b25e714f01f8e2a921fa
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7bd3539150b8b4416c1ef3e8435df0b2b18db72a3b53e80a325b1d948246653b
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b9efd33673a3b22874e09faf8887dab4cf78fa75d3ae16ca5b96616a9ee4f34e
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:414056302bb2c51030907ea288f267bbdafb5b349fed9bbe5df32046e41df462
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b9d2fad89c5a6f1c349b9744f74a33ddbea84929474a3b07af7854b2cb5b948
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:217d507942f28f310e4eeed6bfba8d07fd2bb6dc167c4c4f66eda1c08b6318d6
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:366d6a2b5441cba1b4e3c949688263d18faed9ec24ff34e65646b84a37cd876f
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:06b43a58b72baf558b7e5624a89206c612752404e0e3755fa74aff6a8a856fea
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:398af4c0b1b16509a9dc98b6f538417e0edc1b874f91676dd82e779576f34c02
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ef456277f681db2c07f1ff63be489edd84235ae448b25e8872b7c6ddb3f68afb
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e95cc046a3497bc828055e3c8ba24ab45d17dc68b972f4cff7ec411d034a7f1e
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.48,
6
  "eval_steps": 1000,
7
- "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -956,6 +956,164 @@
956
  "eval_samples_per_second": 16.508,
957
  "eval_steps_per_second": 0.528,
958
  "step": 12000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
959
  }
960
  ],
961
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.56,
6
  "eval_steps": 1000,
7
+ "global_step": 14000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
956
  "eval_samples_per_second": 16.508,
957
  "eval_steps_per_second": 0.528,
958
  "step": 12000
959
+ },
960
+ {
961
+ "epoch": 0.484,
962
+ "grad_norm": 3.012925148010254,
963
+ "learning_rate": 0.0021020354462549986,
964
+ "loss": 43.3377099609375,
965
+ "step": 12100
966
+ },
967
+ {
968
+ "epoch": 0.488,
969
+ "grad_norm": 3.428558826446533,
970
+ "learning_rate": 0.0020769179888207967,
971
+ "loss": 43.5579638671875,
972
+ "step": 12200
973
+ },
974
+ {
975
+ "epoch": 0.492,
976
+ "grad_norm": 3.1122782230377197,
977
+ "learning_rate": 0.0020517883754218,
978
+ "loss": 43.4773828125,
979
+ "step": 12300
980
+ },
981
+ {
982
+ "epoch": 0.496,
983
+ "grad_norm": 2.5048506259918213,
984
+ "learning_rate": 0.0020266505774917446,
985
+ "loss": 43.3037744140625,
986
+ "step": 12400
987
+ },
988
+ {
989
+ "epoch": 0.5,
990
+ "grad_norm": 2.638044595718384,
991
+ "learning_rate": 0.0020015085677578355,
992
+ "loss": 43.562880859375,
993
+ "step": 12500
994
+ },
995
+ {
996
+ "epoch": 0.504,
997
+ "grad_norm": 2.780961036682129,
998
+ "learning_rate": 0.0019763663196129006,
999
+ "loss": 43.3517138671875,
1000
+ "step": 12600
1001
+ },
1002
+ {
1003
+ "epoch": 0.508,
1004
+ "grad_norm": 1.9473482370376587,
1005
+ "learning_rate": 0.0019512278064874483,
1006
+ "loss": 43.109091796875,
1007
+ "step": 12700
1008
+ },
1009
+ {
1010
+ "epoch": 0.512,
1011
+ "grad_norm": 3.174790143966675,
1012
+ "learning_rate": 0.0019260970012217105,
1013
+ "loss": 43.475107421875,
1014
+ "step": 12800
1015
+ },
1016
+ {
1017
+ "epoch": 0.516,
1018
+ "grad_norm": 2.280482769012451,
1019
+ "learning_rate": 0.001900977875437784,
1020
+ "loss": 43.4238671875,
1021
+ "step": 12900
1022
+ },
1023
+ {
1024
+ "epoch": 0.52,
1025
+ "grad_norm": 3.096158266067505,
1026
+ "learning_rate": 0.0018758743989119647,
1027
+ "loss": 43.194453125,
1028
+ "step": 13000
1029
+ },
1030
+ {
1031
+ "epoch": 0.52,
1032
+ "eval_accuracy": 0.21564809384164224,
1033
+ "eval_loss": 43.14116668701172,
1034
+ "eval_runtime": 6.8523,
1035
+ "eval_samples_per_second": 18.242,
1036
+ "eval_steps_per_second": 0.584,
1037
+ "step": 13000
1038
+ },
1039
+ {
1040
+ "epoch": 0.524,
1041
+ "grad_norm": 2.7369933128356934,
1042
+ "learning_rate": 0.0018507905389473706,
1043
+ "loss": 43.233857421875,
1044
+ "step": 13100
1045
+ },
1046
+ {
1047
+ "epoch": 0.528,
1048
+ "grad_norm": 2.2296907901763916,
1049
+ "learning_rate": 0.001825730259746958,
1050
+ "loss": 43.15728515625,
1051
+ "step": 13200
1052
+ },
1053
+ {
1054
+ "epoch": 0.532,
1055
+ "grad_norm": 3.018181324005127,
1056
+ "learning_rate": 0.0018006975217870259,
1057
+ "loss": 43.116005859375,
1058
+ "step": 13300
1059
+ },
1060
+ {
1061
+ "epoch": 0.536,
1062
+ "grad_norm": 2.055466890335083,
1063
+ "learning_rate": 0.0017756962811913122,
1064
+ "loss": 43.1440869140625,
1065
+ "step": 13400
1066
+ },
1067
+ {
1068
+ "epoch": 0.54,
1069
+ "grad_norm": 2.277271032333374,
1070
+ "learning_rate": 0.0017507304891057722,
1071
+ "loss": 43.4723193359375,
1072
+ "step": 13500
1073
+ },
1074
+ {
1075
+ "epoch": 0.544,
1076
+ "grad_norm": 2.3194973468780518,
1077
+ "learning_rate": 0.001725804091074152,
1078
+ "loss": 43.233828125,
1079
+ "step": 13600
1080
+ },
1081
+ {
1082
+ "epoch": 0.548,
1083
+ "grad_norm": 2.0250515937805176,
1084
+ "learning_rate": 0.0017009210264144388,
1085
+ "loss": 43.008115234375,
1086
+ "step": 13700
1087
+ },
1088
+ {
1089
+ "epoch": 0.552,
1090
+ "grad_norm": 2.4250712394714355,
1091
+ "learning_rate": 0.0016760852275963015,
1092
+ "loss": 42.6800048828125,
1093
+ "step": 13800
1094
+ },
1095
+ {
1096
+ "epoch": 0.556,
1097
+ "grad_norm": 1.7238541841506958,
1098
+ "learning_rate": 0.0016513006196196098,
1099
+ "loss": 42.93923828125,
1100
+ "step": 13900
1101
+ },
1102
+ {
1103
+ "epoch": 0.56,
1104
+ "grad_norm": 1.7167351245880127,
1105
+ "learning_rate": 0.0016265711193941361,
1106
+ "loss": 43.267294921875,
1107
+ "step": 14000
1108
+ },
1109
+ {
1110
+ "epoch": 0.56,
1111
+ "eval_accuracy": 0.2188279569892473,
1112
+ "eval_loss": 42.8851318359375,
1113
+ "eval_runtime": 7.0878,
1114
+ "eval_samples_per_second": 17.636,
1115
+ "eval_steps_per_second": 0.564,
1116
+ "step": 14000
1117
  }
1118
  ],
1119
  "logging_steps": 100,