Training in progress, step 18000, checkpoint
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +160 -2
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 386379
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5163584c22b230a285b507e8fe2e44bfe9c5791489c457ec3f5dae50ef6a9d4e
|
| 3 |
size 386379
|
last-checkpoint/pytorch_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1540661735
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ea5c1de12275b49f3c217c8fb0f7b7f68cd81b837676a213fd5cbc5529e41d84
|
| 3 |
size 1540661735
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6d32ac62864aa24d38961f967a353dbdc0c209a83a9a14bd5c6430acf5218f70
|
| 3 |
size 14469
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:af8916051f16387fb69685f7e072b99274ccf08a163a3cca2253ee5f3a7d6975
|
| 3 |
size 14469
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:838f0bffb8a21a934a0dd82575abe5438ad86fa4e56f88f4502576bfc1452b15
|
| 3 |
size 14469
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a4277fe562ff3ebc8bf0029b649400f1d520bbe1f14e9fb0ddeb5516160e4f1e
|
| 3 |
size 14469
|
last-checkpoint/rng_state_4.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2aca493fa7cb3a1e93f1a9b8b666fdc3f1dff09789426df3cc65d5126aa02b10
|
| 3 |
size 14469
|
last-checkpoint/rng_state_5.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:454e7b124610176fbeace81b35b4d3891f06294dd70b8c450a31971fb9d0f6e5
|
| 3 |
size 14469
|
last-checkpoint/rng_state_6.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8a2463c5e875ad508adb42e7029d59e6e8bcbb4d858d11906a1c0881a41dee45
|
| 3 |
size 14469
|
last-checkpoint/rng_state_7.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8f3ee3875e552d5e8fd915d17b82bcf736028026d7ebd8fb029b832116243bc7
|
| 3 |
size 14469
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:218c7072bdbb7a0145270cf1a07d80a739b3014f750ae43e7b4b16e30b41a003
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 1000,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -1272,6 +1272,164 @@
|
|
| 1272 |
"eval_samples_per_second": 17.185,
|
| 1273 |
"eval_steps_per_second": 0.55,
|
| 1274 |
"step": 16000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1275 |
}
|
| 1276 |
],
|
| 1277 |
"logging_steps": 100,
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.72,
|
| 6 |
"eval_steps": 1000,
|
| 7 |
+
"global_step": 18000,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 1272 |
"eval_samples_per_second": 17.185,
|
| 1273 |
"eval_steps_per_second": 0.55,
|
| 1274 |
"step": 16000
|
| 1275 |
+
},
|
| 1276 |
+
{
|
| 1277 |
+
"epoch": 0.644,
|
| 1278 |
+
"grad_norm": 1.2762759923934937,
|
| 1279 |
+
"learning_rate": 0.001126799788845827,
|
| 1280 |
+
"loss": 42.4043359375,
|
| 1281 |
+
"step": 16100
|
| 1282 |
+
},
|
| 1283 |
+
{
|
| 1284 |
+
"epoch": 0.648,
|
| 1285 |
+
"grad_norm": 1.4422513246536255,
|
| 1286 |
+
"learning_rate": 0.0011042495226358784,
|
| 1287 |
+
"loss": 43.033349609375,
|
| 1288 |
+
"step": 16200
|
| 1289 |
+
},
|
| 1290 |
+
{
|
| 1291 |
+
"epoch": 0.652,
|
| 1292 |
+
"grad_norm": 1.816651463508606,
|
| 1293 |
+
"learning_rate": 0.0010818408190361227,
|
| 1294 |
+
"loss": 42.873017578125,
|
| 1295 |
+
"step": 16300
|
| 1296 |
+
},
|
| 1297 |
+
{
|
| 1298 |
+
"epoch": 0.656,
|
| 1299 |
+
"grad_norm": 2.664728879928589,
|
| 1300 |
+
"learning_rate": 0.0010595772194731657,
|
| 1301 |
+
"loss": 42.72642578125,
|
| 1302 |
+
"step": 16400
|
| 1303 |
+
},
|
| 1304 |
+
{
|
| 1305 |
+
"epoch": 0.66,
|
| 1306 |
+
"grad_norm": 2.1422553062438965,
|
| 1307 |
+
"learning_rate": 0.0010374622424416619,
|
| 1308 |
+
"loss": 42.725224609375,
|
| 1309 |
+
"step": 16500
|
| 1310 |
+
},
|
| 1311 |
+
{
|
| 1312 |
+
"epoch": 0.664,
|
| 1313 |
+
"grad_norm": 2.044825315475464,
|
| 1314 |
+
"learning_rate": 0.0010154993829482593,
|
| 1315 |
+
"loss": 42.5387841796875,
|
| 1316 |
+
"step": 16600
|
| 1317 |
+
},
|
| 1318 |
+
{
|
| 1319 |
+
"epoch": 0.668,
|
| 1320 |
+
"grad_norm": 1.400777816772461,
|
| 1321 |
+
"learning_rate": 0.0009936921119592535,
|
| 1322 |
+
"loss": 42.594912109375,
|
| 1323 |
+
"step": 16700
|
| 1324 |
+
},
|
| 1325 |
+
{
|
| 1326 |
+
"epoch": 0.672,
|
| 1327 |
+
"grad_norm": 1.5564935207366943,
|
| 1328 |
+
"learning_rate": 0.0009720438758520462,
|
| 1329 |
+
"loss": 42.6428369140625,
|
| 1330 |
+
"step": 16800
|
| 1331 |
+
},
|
| 1332 |
+
{
|
| 1333 |
+
"epoch": 0.676,
|
| 1334 |
+
"grad_norm": 2.554816961288452,
|
| 1335 |
+
"learning_rate": 0.0009505580958704852,
|
| 1336 |
+
"loss": 42.628232421875,
|
| 1337 |
+
"step": 16900
|
| 1338 |
+
},
|
| 1339 |
+
{
|
| 1340 |
+
"epoch": 0.68,
|
| 1341 |
+
"grad_norm": 1.2824437618255615,
|
| 1342 |
+
"learning_rate": 0.0009292381675841768,
|
| 1343 |
+
"loss": 42.5581103515625,
|
| 1344 |
+
"step": 17000
|
| 1345 |
+
},
|
| 1346 |
+
{
|
| 1347 |
+
"epoch": 0.68,
|
| 1348 |
+
"eval_accuracy": 0.22342913000977518,
|
| 1349 |
+
"eval_loss": 42.39310836791992,
|
| 1350 |
+
"eval_runtime": 7.1782,
|
| 1351 |
+
"eval_samples_per_second": 17.414,
|
| 1352 |
+
"eval_steps_per_second": 0.557,
|
| 1353 |
+
"step": 17000
|
| 1354 |
+
},
|
| 1355 |
+
{
|
| 1356 |
+
"epoch": 0.684,
|
| 1357 |
+
"grad_norm": 1.755340576171875,
|
| 1358 |
+
"learning_rate": 0.0009080874603518585,
|
| 1359 |
+
"loss": 42.472861328125,
|
| 1360 |
+
"step": 17100
|
| 1361 |
+
},
|
| 1362 |
+
{
|
| 1363 |
+
"epoch": 0.688,
|
| 1364 |
+
"grad_norm": 1.1157244443893433,
|
| 1365 |
+
"learning_rate": 0.0008871093167889121,
|
| 1366 |
+
"loss": 42.5861279296875,
|
| 1367 |
+
"step": 17200
|
| 1368 |
+
},
|
| 1369 |
+
{
|
| 1370 |
+
"epoch": 0.692,
|
| 1371 |
+
"grad_norm": 1.2528804540634155,
|
| 1372 |
+
"learning_rate": 0.0008663070522391008,
|
| 1373 |
+
"loss": 42.4558740234375,
|
| 1374 |
+
"step": 17300
|
| 1375 |
+
},
|
| 1376 |
+
{
|
| 1377 |
+
"epoch": 0.696,
|
| 1378 |
+
"grad_norm": 1.7521710395812988,
|
| 1379 |
+
"learning_rate": 0.0008456839542506229,
|
| 1380 |
+
"loss": 42.712470703125,
|
| 1381 |
+
"step": 17400
|
| 1382 |
+
},
|
| 1383 |
+
{
|
| 1384 |
+
"epoch": 0.7,
|
| 1385 |
+
"grad_norm": 1.830307960510254,
|
| 1386 |
+
"learning_rate": 0.0008252432820565525,
|
| 1387 |
+
"loss": 42.67123046875,
|
| 1388 |
+
"step": 17500
|
| 1389 |
+
},
|
| 1390 |
+
{
|
| 1391 |
+
"epoch": 0.704,
|
| 1392 |
+
"grad_norm": 1.6736066341400146,
|
| 1393 |
+
"learning_rate": 0.0008049882660597558,
|
| 1394 |
+
"loss": 42.42759765625,
|
| 1395 |
+
"step": 17600
|
| 1396 |
+
},
|
| 1397 |
+
{
|
| 1398 |
+
"epoch": 0.708,
|
| 1399 |
+
"grad_norm": 1.5861881971359253,
|
| 1400 |
+
"learning_rate": 0.0007849221073223665,
|
| 1401 |
+
"loss": 42.5102734375,
|
| 1402 |
+
"step": 17700
|
| 1403 |
+
},
|
| 1404 |
+
{
|
| 1405 |
+
"epoch": 0.712,
|
| 1406 |
+
"grad_norm": 1.659502387046814,
|
| 1407 |
+
"learning_rate": 0.0007650479770598948,
|
| 1408 |
+
"loss": 42.2067578125,
|
| 1409 |
+
"step": 17800
|
| 1410 |
+
},
|
| 1411 |
+
{
|
| 1412 |
+
"epoch": 0.716,
|
| 1413 |
+
"grad_norm": 1.2964304685592651,
|
| 1414 |
+
"learning_rate": 0.0007453690161400562,
|
| 1415 |
+
"loss": 42.520947265625,
|
| 1416 |
+
"step": 17900
|
| 1417 |
+
},
|
| 1418 |
+
{
|
| 1419 |
+
"epoch": 0.72,
|
| 1420 |
+
"grad_norm": 1.7349175214767456,
|
| 1421 |
+
"learning_rate": 0.000725888334586394,
|
| 1422 |
+
"loss": 42.7132568359375,
|
| 1423 |
+
"step": 18000
|
| 1424 |
+
},
|
| 1425 |
+
{
|
| 1426 |
+
"epoch": 0.72,
|
| 1427 |
+
"eval_accuracy": 0.224316715542522,
|
| 1428 |
+
"eval_loss": 42.29633331298828,
|
| 1429 |
+
"eval_runtime": 7.2524,
|
| 1430 |
+
"eval_samples_per_second": 17.236,
|
| 1431 |
+
"eval_steps_per_second": 0.552,
|
| 1432 |
+
"step": 18000
|
| 1433 |
}
|
| 1434 |
],
|
| 1435 |
"logging_steps": 100,
|