Training in progress, step 3300, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 943004768
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3c5e3e5b6e9a9841f3c38ec65ab929aac3efcce069fdaeb1d1a933cedef6d197
|
| 3 |
size 943004768
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1807338490
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:811ca9cb0c040c15f78f49bc0f526fee796e613b88cc6427995ea0b72bc02162
|
| 3 |
size 1807338490
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3b67164c1bb8be00156b461f1da94ff5ae272c2934b635de5082bdb2afa9539b
|
| 3 |
size 14244
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1256
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aa72ca4314eb7860ade352ab52d28713bfc1b286396d413d45893a04bd168889
|
| 3 |
size 1256
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
"best_metric": 0.006907718721777201,
|
| 3 |
"best_model_checkpoint": "./output/checkpoint-2400",
|
| 4 |
-
"epoch": 0.
|
| 5 |
"eval_steps": 150,
|
| 6 |
-
"global_step":
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
@@ -2380,6 +2380,119 @@
|
|
| 2380 |
"eval_samples_per_second": 9.753,
|
| 2381 |
"eval_steps_per_second": 9.753,
|
| 2382 |
"step": 3150
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2383 |
}
|
| 2384 |
],
|
| 2385 |
"logging_steps": 10,
|
|
@@ -2399,7 +2512,7 @@
|
|
| 2399 |
"attributes": {}
|
| 2400 |
}
|
| 2401 |
},
|
| 2402 |
-
"total_flos": 2.
|
| 2403 |
"train_batch_size": 4,
|
| 2404 |
"trial_name": null,
|
| 2405 |
"trial_params": null
|
|
|
|
| 1 |
{
|
| 2 |
"best_metric": 0.006907718721777201,
|
| 3 |
"best_model_checkpoint": "./output/checkpoint-2400",
|
| 4 |
+
"epoch": 0.5519317611640743,
|
| 5 |
"eval_steps": 150,
|
| 6 |
+
"global_step": 3300,
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
|
|
| 2380 |
"eval_samples_per_second": 9.753,
|
| 2381 |
"eval_steps_per_second": 9.753,
|
| 2382 |
"step": 3150
|
| 2383 |
+
},
|
| 2384 |
+
{
|
| 2385 |
+
"epoch": 0.5285164743268105,
|
| 2386 |
+
"grad_norm": 0.00785563699901104,
|
| 2387 |
+
"learning_rate": 1.3613553844641483e-05,
|
| 2388 |
+
"loss": 0.0004,
|
| 2389 |
+
"step": 3160
|
| 2390 |
+
},
|
| 2391 |
+
{
|
| 2392 |
+
"epoch": 0.5301889948151864,
|
| 2393 |
+
"grad_norm": 0.0006820244598202407,
|
| 2394 |
+
"learning_rate": 1.3483326530583184e-05,
|
| 2395 |
+
"loss": 0.0005,
|
| 2396 |
+
"step": 3170
|
| 2397 |
+
},
|
| 2398 |
+
{
|
| 2399 |
+
"epoch": 0.5318615153035625,
|
| 2400 |
+
"grad_norm": 0.006723721977323294,
|
| 2401 |
+
"learning_rate": 1.3353449303613682e-05,
|
| 2402 |
+
"loss": 0.0003,
|
| 2403 |
+
"step": 3180
|
| 2404 |
+
},
|
| 2405 |
+
{
|
| 2406 |
+
"epoch": 0.5335340357919385,
|
| 2407 |
+
"grad_norm": 0.00332651031203568,
|
| 2408 |
+
"learning_rate": 1.3223927502477084e-05,
|
| 2409 |
+
"loss": 0.0001,
|
| 2410 |
+
"step": 3190
|
| 2411 |
+
},
|
| 2412 |
+
{
|
| 2413 |
+
"epoch": 0.5352065562803144,
|
| 2414 |
+
"grad_norm": 0.002680362667888403,
|
| 2415 |
+
"learning_rate": 1.3094766451307336e-05,
|
| 2416 |
+
"loss": 0.0003,
|
| 2417 |
+
"step": 3200
|
| 2418 |
+
},
|
| 2419 |
+
{
|
| 2420 |
+
"epoch": 0.5368790767686904,
|
| 2421 |
+
"grad_norm": 0.0007611092296428978,
|
| 2422 |
+
"learning_rate": 1.2965971459409366e-05,
|
| 2423 |
+
"loss": 0.0007,
|
| 2424 |
+
"step": 3210
|
| 2425 |
+
},
|
| 2426 |
+
{
|
| 2427 |
+
"epoch": 0.5385515972570664,
|
| 2428 |
+
"grad_norm": 0.03528850898146629,
|
| 2429 |
+
"learning_rate": 1.2837547821040825e-05,
|
| 2430 |
+
"loss": 0.0034,
|
| 2431 |
+
"step": 3220
|
| 2432 |
+
},
|
| 2433 |
+
{
|
| 2434 |
+
"epoch": 0.5402241177454424,
|
| 2435 |
+
"grad_norm": 0.02707619220018387,
|
| 2436 |
+
"learning_rate": 1.2709500815194487e-05,
|
| 2437 |
+
"loss": 0.0129,
|
| 2438 |
+
"step": 3230
|
| 2439 |
+
},
|
| 2440 |
+
{
|
| 2441 |
+
"epoch": 0.5418966382338184,
|
| 2442 |
+
"grad_norm": 0.4840797185897827,
|
| 2443 |
+
"learning_rate": 1.2581835705381243e-05,
|
| 2444 |
+
"loss": 0.0046,
|
| 2445 |
+
"step": 3240
|
| 2446 |
+
},
|
| 2447 |
+
{
|
| 2448 |
+
"epoch": 0.5435691587221944,
|
| 2449 |
+
"grad_norm": 0.027722898870706558,
|
| 2450 |
+
"learning_rate": 1.2454557739413722e-05,
|
| 2451 |
+
"loss": 0.0003,
|
| 2452 |
+
"step": 3250
|
| 2453 |
+
},
|
| 2454 |
+
{
|
| 2455 |
+
"epoch": 0.5452416792105703,
|
| 2456 |
+
"grad_norm": 0.0002542615111451596,
|
| 2457 |
+
"learning_rate": 1.2327672149190595e-05,
|
| 2458 |
+
"loss": 0.0002,
|
| 2459 |
+
"step": 3260
|
| 2460 |
+
},
|
| 2461 |
+
{
|
| 2462 |
+
"epoch": 0.5469141996989463,
|
| 2463 |
+
"grad_norm": 0.29150494933128357,
|
| 2464 |
+
"learning_rate": 1.2201184150481497e-05,
|
| 2465 |
+
"loss": 0.006,
|
| 2466 |
+
"step": 3270
|
| 2467 |
+
},
|
| 2468 |
+
{
|
| 2469 |
+
"epoch": 0.5485867201873222,
|
| 2470 |
+
"grad_norm": 0.01146694552153349,
|
| 2471 |
+
"learning_rate": 1.2075098942712635e-05,
|
| 2472 |
+
"loss": 0.001,
|
| 2473 |
+
"step": 3280
|
| 2474 |
+
},
|
| 2475 |
+
{
|
| 2476 |
+
"epoch": 0.5502592406756983,
|
| 2477 |
+
"grad_norm": 0.15940113365650177,
|
| 2478 |
+
"learning_rate": 1.1949421708753062e-05,
|
| 2479 |
+
"loss": 0.016,
|
| 2480 |
+
"step": 3290
|
| 2481 |
+
},
|
| 2482 |
+
{
|
| 2483 |
+
"epoch": 0.5519317611640743,
|
| 2484 |
+
"grad_norm": 0.000170723840710707,
|
| 2485 |
+
"learning_rate": 1.1824157614701629e-05,
|
| 2486 |
+
"loss": 0.0007,
|
| 2487 |
+
"step": 3300
|
| 2488 |
+
},
|
| 2489 |
+
{
|
| 2490 |
+
"epoch": 0.5519317611640743,
|
| 2491 |
+
"eval_loss": 0.00781547836959362,
|
| 2492 |
+
"eval_runtime": 57.4069,
|
| 2493 |
+
"eval_samples_per_second": 8.727,
|
| 2494 |
+
"eval_steps_per_second": 8.727,
|
| 2495 |
+
"step": 3300
|
| 2496 |
}
|
| 2497 |
],
|
| 2498 |
"logging_steps": 10,
|
|
|
|
| 2512 |
"attributes": {}
|
| 2513 |
}
|
| 2514 |
},
|
| 2515 |
+
"total_flos": 2.1405275369668608e+17,
|
| 2516 |
"train_batch_size": 4,
|
| 2517 |
"trial_name": null,
|
| 2518 |
"trial_params": null
|