Training in progress, step 2150, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 479005064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6b6cad65a1239e0e03962e7f4ef559abd941989cf40db0a80d0189112e42eb72
|
| 3 |
size 479005064
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 958299770
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2b8a70c19523d8c6e08995eab7545028af88fed635908dfc05a7946675d3475a
|
| 3 |
size 958299770
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f9759af3432ea363e3c64dd28f0228b81e57792402129de6cb79a2e5033a37e1
|
| 3 |
size 14244
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2e6aaab334f3a9481d30ef9495cbe830ba9fac2b9efdeee34a635582fea7ec27
|
| 3 |
size 1064
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
"best_metric": null,
|
| 3 |
"best_model_checkpoint": null,
|
| 4 |
-
"epoch": 0.
|
| 5 |
"eval_steps": 500,
|
| 6 |
-
"global_step":
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
@@ -2551,6 +2551,42 @@
|
|
| 2551 |
"reward_std": 0.307485481351614,
|
| 2552 |
"rewards/custom_reward_simplified_v7_dblog": 0.7875,
|
| 2553 |
"step": 2120
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2554 |
}
|
| 2555 |
],
|
| 2556 |
"logging_steps": 10,
|
|
|
|
| 1 |
{
|
| 2 |
"best_metric": null,
|
| 3 |
"best_model_checkpoint": null,
|
| 4 |
+
"epoch": 0.017123151296979156,
|
| 5 |
"eval_steps": 500,
|
| 6 |
+
"global_step": 2150,
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
|
|
| 2551 |
"reward_std": 0.307485481351614,
|
| 2552 |
"rewards/custom_reward_simplified_v7_dblog": 0.7875,
|
| 2553 |
"step": 2120
|
| 2554 |
+
},
|
| 2555 |
+
{
|
| 2556 |
+
"completion_length": 685.39375,
|
| 2557 |
+
"epoch": 0.016963866168635166,
|
| 2558 |
+
"grad_norm": 0.30204537510871887,
|
| 2559 |
+
"kl": 0.018967814440838993,
|
| 2560 |
+
"learning_rate": 1.9030116872178317e-07,
|
| 2561 |
+
"loss": 0.0008,
|
| 2562 |
+
"reward": 0.803125,
|
| 2563 |
+
"reward_std": 0.3279333204030991,
|
| 2564 |
+
"rewards/custom_reward_simplified_v7_dblog": 0.803125,
|
| 2565 |
+
"step": 2130
|
| 2566 |
+
},
|
| 2567 |
+
{
|
| 2568 |
+
"completion_length": 674.49375,
|
| 2569 |
+
"epoch": 0.01704350873280716,
|
| 2570 |
+
"grad_norm": 0.012012571096420288,
|
| 2571 |
+
"kl": 0.02170075795147568,
|
| 2572 |
+
"learning_rate": 1.7663118943294367e-07,
|
| 2573 |
+
"loss": 0.0009,
|
| 2574 |
+
"reward": 0.703125,
|
| 2575 |
+
"reward_std": 0.2257047951221466,
|
| 2576 |
+
"rewards/custom_reward_simplified_v7_dblog": 0.703125,
|
| 2577 |
+
"step": 2140
|
| 2578 |
+
},
|
| 2579 |
+
{
|
| 2580 |
+
"completion_length": 694.63125,
|
| 2581 |
+
"epoch": 0.017123151296979156,
|
| 2582 |
+
"grad_norm": 0.01635037176311016,
|
| 2583 |
+
"kl": 0.02094450539443642,
|
| 2584 |
+
"learning_rate": 1.6345268662752904e-07,
|
| 2585 |
+
"loss": 0.0008,
|
| 2586 |
+
"reward": 0.7125,
|
| 2587 |
+
"reward_std": 0.2917635254561901,
|
| 2588 |
+
"rewards/custom_reward_simplified_v7_dblog": 0.7125,
|
| 2589 |
+
"step": 2150
|
| 2590 |
}
|
| 2591 |
],
|
| 2592 |
"logging_steps": 10,
|