Training in progress, step 3000, checkpoint
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +91 -3
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 386379
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3dbc2fb0a6c5f2970cd53453e764a649ac6607abbf79cfa4749975093698eaf6
|
| 3 |
size 386379
|
last-checkpoint/pytorch_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1540661735
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e6a587f037c6fe6673f8514fdde6c05e7562af204e86df7c654d7dae4b255f1f
|
| 3 |
size 1540661735
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:218b503b380a0e1262fb31f7e6ad9fa8d4cf2f109ff7eda4aa5eed8f7aef324c
|
| 3 |
size 14469
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:97a5947f809acf4e5bfead2eebc83edaff5624e9d14837c9fa503560db15c663
|
| 3 |
size 14469
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0c7885e48aad56e6bb142603863a10c62d22f7a2518b4baba80b8d015b666e48
|
| 3 |
size 14469
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c7610f49093646d2e22293764225fe9b8113bf86c1e3e290ea735ac5e1bd6ae7
|
| 3 |
size 14469
|
last-checkpoint/rng_state_4.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9ca1bdf5df09ff68c66a2f5cdc44a65ad89d1674b79918065e22e92ce4210da0
|
| 3 |
size 14469
|
last-checkpoint/rng_state_5.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:112e61e6b9d44ab507241b2ef32d6cfa5970d93605a3430d819c9d6544dddd34
|
| 3 |
size 14469
|
last-checkpoint/rng_state_6.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2dc8f42a650b40b2f2252aaa98d697c4c891fd4cd9d7cc47c1c2b9764ba09934
|
| 3 |
size 14469
|
last-checkpoint/rng_state_7.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f8e75193fe6e093d812dbc12982db192f141515d07b4fe6ba85ffc695a84fb99
|
| 3 |
size 14469
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b36204b8c6666543761ec77ac720f6c626682baa52cec8cf26572629d35d9b79
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch":
|
| 6 |
"eval_steps": 250,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -448,6 +448,94 @@
|
|
| 448 |
"eval_samples_per_second": 18.128,
|
| 449 |
"eval_steps_per_second": 0.58,
|
| 450 |
"step": 2500
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 451 |
}
|
| 452 |
],
|
| 453 |
"logging_steps": 50,
|
|
@@ -462,7 +550,7 @@
|
|
| 462 |
"should_evaluate": false,
|
| 463 |
"should_log": false,
|
| 464 |
"should_save": true,
|
| 465 |
-
"should_training_stop":
|
| 466 |
},
|
| 467 |
"attributes": {}
|
| 468 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 1.0,
|
| 6 |
"eval_steps": 250,
|
| 7 |
+
"global_step": 3000,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 448 |
"eval_samples_per_second": 18.128,
|
| 449 |
"eval_steps_per_second": 0.58,
|
| 450 |
"step": 2500
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"epoch": 0.85,
|
| 454 |
+
"grad_norm": 455.0086975097656,
|
| 455 |
+
"learning_rate": 2.4210285221698124e-05,
|
| 456 |
+
"loss": 1739.21203125,
|
| 457 |
+
"step": 2550
|
| 458 |
+
},
|
| 459 |
+
{
|
| 460 |
+
"epoch": 0.8666666666666667,
|
| 461 |
+
"grad_norm": 460.4869079589844,
|
| 462 |
+
"learning_rate": 1.922276599001802e-05,
|
| 463 |
+
"loss": 1737.5628125,
|
| 464 |
+
"step": 2600
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"epoch": 0.8833333333333333,
|
| 468 |
+
"grad_norm": 488.4925231933594,
|
| 469 |
+
"learning_rate": 1.4784261276487333e-05,
|
| 470 |
+
"loss": 1737.22265625,
|
| 471 |
+
"step": 2650
|
| 472 |
+
},
|
| 473 |
+
{
|
| 474 |
+
"epoch": 0.9,
|
| 475 |
+
"grad_norm": 602.5552978515625,
|
| 476 |
+
"learning_rate": 1.0908250674042331e-05,
|
| 477 |
+
"loss": 1736.81359375,
|
| 478 |
+
"step": 2700
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"epoch": 0.9166666666666666,
|
| 482 |
+
"grad_norm": 375.4463195800781,
|
| 483 |
+
"learning_rate": 7.606505499491246e-06,
|
| 484 |
+
"loss": 1736.9728125,
|
| 485 |
+
"step": 2750
|
| 486 |
+
},
|
| 487 |
+
{
|
| 488 |
+
"epoch": 0.9166666666666666,
|
| 489 |
+
"eval_accuracy": 0.17933235581622678,
|
| 490 |
+
"eval_loss": 57.458404541015625,
|
| 491 |
+
"eval_runtime": 7.136,
|
| 492 |
+
"eval_samples_per_second": 17.517,
|
| 493 |
+
"eval_steps_per_second": 0.561,
|
| 494 |
+
"step": 2750
|
| 495 |
+
},
|
| 496 |
+
{
|
| 497 |
+
"epoch": 0.9333333333333333,
|
| 498 |
+
"grad_norm": 411.4505920410156,
|
| 499 |
+
"learning_rate": 4.88905304441194e-06,
|
| 500 |
+
"loss": 1736.17953125,
|
| 501 |
+
"step": 2800
|
| 502 |
+
},
|
| 503 |
+
{
|
| 504 |
+
"epoch": 0.95,
|
| 505 |
+
"grad_norm": 255.17337036132812,
|
| 506 |
+
"learning_rate": 2.7641461225969665e-06,
|
| 507 |
+
"loss": 1737.8128125,
|
| 508 |
+
"step": 2850
|
| 509 |
+
},
|
| 510 |
+
{
|
| 511 |
+
"epoch": 0.9666666666666667,
|
| 512 |
+
"grad_norm": 709.8735961914062,
|
| 513 |
+
"learning_rate": 1.2382380065293576e-06,
|
| 514 |
+
"loss": 1737.474375,
|
| 515 |
+
"step": 2900
|
| 516 |
+
},
|
| 517 |
+
{
|
| 518 |
+
"epoch": 0.9833333333333333,
|
| 519 |
+
"grad_norm": 485.8822326660156,
|
| 520 |
+
"learning_rate": 3.159628290061445e-07,
|
| 521 |
+
"loss": 1737.075,
|
| 522 |
+
"step": 2950
|
| 523 |
+
},
|
| 524 |
+
{
|
| 525 |
+
"epoch": 1.0,
|
| 526 |
+
"grad_norm": 478.07696533203125,
|
| 527 |
+
"learning_rate": 1.2150942938493615e-10,
|
| 528 |
+
"loss": 1737.3753125,
|
| 529 |
+
"step": 3000
|
| 530 |
+
},
|
| 531 |
+
{
|
| 532 |
+
"epoch": 1.0,
|
| 533 |
+
"eval_accuracy": 0.17936754643206257,
|
| 534 |
+
"eval_loss": 57.4544677734375,
|
| 535 |
+
"eval_runtime": 7.3114,
|
| 536 |
+
"eval_samples_per_second": 17.097,
|
| 537 |
+
"eval_steps_per_second": 0.547,
|
| 538 |
+
"step": 3000
|
| 539 |
}
|
| 540 |
],
|
| 541 |
"logging_steps": 50,
|
|
|
|
| 550 |
"should_evaluate": false,
|
| 551 |
"should_log": false,
|
| 552 |
"should_save": true,
|
| 553 |
+
"should_training_stop": true
|
| 554 |
},
|
| 555 |
"attributes": {}
|
| 556 |
}
|