Training in progress, step 2000, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 479005064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7d841bdf6b4728ba25ddfa075a6ecf7ffcd91ad64f151f5984cfdb0fb36616e2
|
| 3 |
size 479005064
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 958299770
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b591aa40759fbc489d41cede5b6d509cefae3af7c306b30fa1e6a7a4b8ec4837
|
| 3 |
size 958299770
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b9ffa2392d644ba2690b0835df4eac79e599506fd693b988dfb49d247c7e500c
|
| 3 |
size 14244
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0119210f3709b267b1dcdc2165f2b55aac98c420d5275ee5428e502b1f632094
|
| 3 |
size 1064
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
"best_metric": null,
|
| 3 |
"best_model_checkpoint": null,
|
| 4 |
-
"epoch": 0.
|
| 5 |
"eval_steps": 500,
|
| 6 |
-
"global_step":
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
@@ -2371,6 +2371,42 @@
|
|
| 2371 |
"reward_std": 0.2198973834514618,
|
| 2372 |
"rewards/custom_reward_simplified_v7_dblog": 0.715625,
|
| 2373 |
"step": 1970
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2374 |
}
|
| 2375 |
],
|
| 2376 |
"logging_steps": 10,
|
|
|
|
| 1 |
{
|
| 2 |
"best_metric": null,
|
| 3 |
"best_model_checkpoint": null,
|
| 4 |
+
"epoch": 0.015928512834399215,
|
| 5 |
"eval_steps": 500,
|
| 6 |
+
"global_step": 2000,
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
|
|
| 2371 |
"reward_std": 0.2198973834514618,
|
| 2372 |
"rewards/custom_reward_simplified_v7_dblog": 0.715625,
|
| 2373 |
"step": 1970
|
| 2374 |
+
},
|
| 2375 |
+
{
|
| 2376 |
+
"completion_length": 641.45,
|
| 2377 |
+
"epoch": 0.015769227706055225,
|
| 2378 |
+
"grad_norm": 0.23187489807605743,
|
| 2379 |
+
"kl": 0.017989515024237335,
|
| 2380 |
+
"learning_rate": 4.5211988927752026e-07,
|
| 2381 |
+
"loss": 0.0007,
|
| 2382 |
+
"reward": 0.7875,
|
| 2383 |
+
"reward_std": 0.24450960606336594,
|
| 2384 |
+
"rewards/custom_reward_simplified_v7_dblog": 0.7875,
|
| 2385 |
+
"step": 1980
|
| 2386 |
+
},
|
| 2387 |
+
{
|
| 2388 |
+
"completion_length": 643.6375,
|
| 2389 |
+
"epoch": 0.01584887027022722,
|
| 2390 |
+
"grad_norm": 0.235895574092865,
|
| 2391 |
+
"kl": 0.015841626143082977,
|
| 2392 |
+
"learning_rate": 4.3148139714622365e-07,
|
| 2393 |
+
"loss": 0.0006,
|
| 2394 |
+
"reward": 0.896875,
|
| 2395 |
+
"reward_std": 0.26189937368035315,
|
| 2396 |
+
"rewards/custom_reward_simplified_v7_dblog": 0.896875,
|
| 2397 |
+
"step": 1990
|
| 2398 |
+
},
|
| 2399 |
+
{
|
| 2400 |
+
"completion_length": 629.60625,
|
| 2401 |
+
"epoch": 0.015928512834399215,
|
| 2402 |
+
"grad_norm": 0.2776155471801758,
|
| 2403 |
+
"kl": 0.015184593386948109,
|
| 2404 |
+
"learning_rate": 4.1128047146765936e-07,
|
| 2405 |
+
"loss": 0.0006,
|
| 2406 |
+
"reward": 0.921875,
|
| 2407 |
+
"reward_std": 0.23378355875611306,
|
| 2408 |
+
"rewards/custom_reward_simplified_v7_dblog": 0.921875,
|
| 2409 |
+
"step": 2000
|
| 2410 |
}
|
| 2411 |
],
|
| 2412 |
"logging_steps": 10,
|