CodeIsAbstract commited on
Commit
d21fa02
·
verified ·
1 Parent(s): 6fdfa80

Training in progress, step 16000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:75ba0b7a63e25df56e29b8b1f4ae481007e80d1cbde28fce2cb4ed43b8b03d7f
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:47d722dd48084253e30edcdf2b9c7c11a74b1b8db5e9fdd03e44b6b8722f9fe4
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bbeaa2e3e7f8d9416a2aff04c3f41cb4a4fbb000165b5f4b0f96556dc4f75345
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cc6390d79743c341e54409953964ed7435ac3ae79f72a9c0b2a61d5bfa4014e
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b9fd33b0410b29f1c36e49a77f8193ae40be73acb6bc07f4d04bbf2f5a48485d
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0dcdb042e82db2feeaf162fa2c2cb05728354faf3a44aadef50aea6b23b1d50f
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3b178b887d0edcf498486b28f0a4fa9660c9ddb3e2c9439c66b45fe0ea200cf9
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2e4e7d621db7965ac0d0149ed05a3a6f4a8f4d9ee38752c9f0d6d1b7c145e73
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3ffac77d94e352efd69a7a08457843666ef6e11840940a75b79aa309c60b093e
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:508d87a75853c2525702937f0b29e20460fd10ed4559a13863d962226c0b64f6
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a4793c6e47d0cc17f59428134402f504910de29559f3b25e714f01f8e2a921fa
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e58646a1d294f0200655cb40f9e0932383e71653f6e1881d789b0865a448d379
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b9efd33673a3b22874e09faf8887dab4cf78fa75d3ae16ca5b96616a9ee4f34e
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12049670dc229efa48859b2adaa7bab30026005b65de953934912e8618042ff1
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7b9d2fad89c5a6f1c349b9744f74a33ddbea84929474a3b07af7854b2cb5b948
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73cb0f9f2044cff462979fc73428559a94af22b0f0ccb7178a43a2c54c6bd489
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:366d6a2b5441cba1b4e3c949688263d18faed9ec24ff34e65646b84a37cd876f
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d5d3e29a67a07d0732605625beeef4e37c068d24accf9b8ee268df48cbfb581
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:398af4c0b1b16509a9dc98b6f538417e0edc1b874f91676dd82e779576f34c02
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a884f7fff4609aa43572eb70b86f31ae319b33f5de3caf0a4442a33602de76c4
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e95cc046a3497bc828055e3c8ba24ab45d17dc68b972f4cff7ec411d034a7f1e
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5763ed75e068d243df01e54fda7497cd86e57ed3e9ed4dadca20a2d87ccb859e
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.56,
6
  "eval_steps": 1000,
7
- "global_step": 14000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -1114,6 +1114,164 @@
1114
  "eval_samples_per_second": 17.636,
1115
  "eval_steps_per_second": 0.564,
1116
  "step": 14000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1117
  }
1118
  ],
1119
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.64,
6
  "eval_steps": 1000,
7
+ "global_step": 16000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
1114
  "eval_samples_per_second": 17.636,
1115
  "eval_steps_per_second": 0.564,
1116
  "step": 14000
1117
+ },
1118
+ {
1119
+ "epoch": 0.564,
1120
+ "grad_norm": 2.029331684112549,
1121
+ "learning_rate": 0.0016019006351205318,
1122
+ "loss": 43.2043701171875,
1123
+ "step": 14100
1124
+ },
1125
+ {
1126
+ "epoch": 0.568,
1127
+ "grad_norm": 2.470975399017334,
1128
+ "learning_rate": 0.0015772930656726884,
1129
+ "loss": 43.1994921875,
1130
+ "step": 14200
1131
+ },
1132
+ {
1133
+ "epoch": 0.572,
1134
+ "grad_norm": 1.9896589517593384,
1135
+ "learning_rate": 0.0015527522999815634,
1136
+ "loss": 42.8731396484375,
1137
+ "step": 14300
1138
+ },
1139
+ {
1140
+ "epoch": 0.576,
1141
+ "grad_norm": 2.9005286693573,
1142
+ "learning_rate": 0.001528282216420582,
1143
+ "loss": 42.8190380859375,
1144
+ "step": 14400
1145
+ },
1146
+ {
1147
+ "epoch": 0.58,
1148
+ "grad_norm": 2.300215244293213,
1149
+ "learning_rate": 0.0015038866821927073,
1150
+ "loss": 42.9506982421875,
1151
+ "step": 14500
1152
+ },
1153
+ {
1154
+ "epoch": 0.584,
1155
+ "grad_norm": 1.927196979522705,
1156
+ "learning_rate": 0.0014795695527192762,
1157
+ "loss": 42.97125,
1158
+ "step": 14600
1159
+ },
1160
+ {
1161
+ "epoch": 0.588,
1162
+ "grad_norm": 1.9869160652160645,
1163
+ "learning_rate": 0.001455334671030695,
1164
+ "loss": 43.0368505859375,
1165
+ "step": 14700
1166
+ },
1167
+ {
1168
+ "epoch": 0.592,
1169
+ "grad_norm": 2.05895733833313,
1170
+ "learning_rate": 0.0014311858671590934,
1171
+ "loss": 42.886181640625,
1172
+ "step": 14800
1173
+ },
1174
+ {
1175
+ "epoch": 0.596,
1176
+ "grad_norm": 2.5150656700134277,
1177
+ "learning_rate": 0.0014071269575330373,
1178
+ "loss": 42.9019677734375,
1179
+ "step": 14900
1180
+ },
1181
+ {
1182
+ "epoch": 0.6,
1183
+ "grad_norm": 2.3601467609405518,
1184
+ "learning_rate": 0.0013831617443743852,
1185
+ "loss": 42.6784130859375,
1186
+ "step": 15000
1187
+ },
1188
+ {
1189
+ "epoch": 0.6,
1190
+ "eval_accuracy": 0.22081524926686216,
1191
+ "eval_loss": 42.696502685546875,
1192
+ "eval_runtime": 7.2547,
1193
+ "eval_samples_per_second": 17.23,
1194
+ "eval_steps_per_second": 0.551,
1195
+ "step": 15000
1196
+ },
1197
+ {
1198
+ "epoch": 0.604,
1199
+ "grad_norm": 2.1606075763702393,
1200
+ "learning_rate": 0.0013592940150973943,
1201
+ "loss": 42.9130859375,
1202
+ "step": 15100
1203
+ },
1204
+ {
1205
+ "epoch": 0.608,
1206
+ "grad_norm": 2.021022319793701,
1207
+ "learning_rate": 0.0013355275417101637,
1208
+ "loss": 42.95048828125,
1209
+ "step": 15200
1210
+ },
1211
+ {
1212
+ "epoch": 0.612,
1213
+ "grad_norm": 2.6041364669799805,
1214
+ "learning_rate": 0.0013118660802185161,
1215
+ "loss": 42.6790380859375,
1216
+ "step": 15300
1217
+ },
1218
+ {
1219
+ "epoch": 0.616,
1220
+ "grad_norm": 2.121518135070801,
1221
+ "learning_rate": 0.0012883133700324026,
1222
+ "loss": 42.72169921875,
1223
+ "step": 15400
1224
+ },
1225
+ {
1226
+ "epoch": 0.62,
1227
+ "grad_norm": 2.0749258995056152,
1228
+ "learning_rate": 0.0012648731333749373,
1229
+ "loss": 42.6244970703125,
1230
+ "step": 15500
1231
+ },
1232
+ {
1233
+ "epoch": 0.624,
1234
+ "grad_norm": 2.3933210372924805,
1235
+ "learning_rate": 0.0012415490746941434,
1236
+ "loss": 42.629208984375,
1237
+ "step": 15600
1238
+ },
1239
+ {
1240
+ "epoch": 0.628,
1241
+ "grad_norm": 2.163004159927368,
1242
+ "learning_rate": 0.0012183448800775084,
1243
+ "loss": 42.9462548828125,
1244
+ "step": 15700
1245
+ },
1246
+ {
1247
+ "epoch": 0.632,
1248
+ "grad_norm": 1.9806509017944336,
1249
+ "learning_rate": 0.0011952642166694445,
1250
+ "loss": 42.9928662109375,
1251
+ "step": 15800
1252
+ },
1253
+ {
1254
+ "epoch": 0.636,
1255
+ "grad_norm": 2.2631726264953613,
1256
+ "learning_rate": 0.0011723107320917396,
1257
+ "loss": 42.7688671875,
1258
+ "step": 15900
1259
+ },
1260
+ {
1261
+ "epoch": 0.64,
1262
+ "grad_norm": 2.2971158027648926,
1263
+ "learning_rate": 0.0011494880538670915,
1264
+ "loss": 42.37572265625,
1265
+ "step": 16000
1266
+ },
1267
+ {
1268
+ "epoch": 0.64,
1269
+ "eval_accuracy": 0.2221368523949169,
1270
+ "eval_loss": 42.538299560546875,
1271
+ "eval_runtime": 7.2739,
1272
+ "eval_samples_per_second": 17.185,
1273
+ "eval_steps_per_second": 0.55,
1274
+ "step": 16000
1275
  }
1276
  ],
1277
  "logging_steps": 100,