{ "best_global_step": 366, "best_metric": 0.41911664605140686, "best_model_checkpoint": "training_output/run_20260713_203733/model/checkpoint-366", "epoch": 1.0, "eval_steps": 500, "global_step": 366, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0273224043715847, "grad_norm": 4.927176475524902, "learning_rate": 7.377049180327868e-07, "loss": 0.6899909496307373, "step": 10 }, { "epoch": 0.0546448087431694, "grad_norm": 6.0230512619018555, "learning_rate": 1.557377049180328e-06, "loss": 0.7346426486968994, "step": 20 }, { "epoch": 0.08196721311475409, "grad_norm": 6.2535271644592285, "learning_rate": 2.377049180327869e-06, "loss": 0.6693419456481934, "step": 30 }, { "epoch": 0.1092896174863388, "grad_norm": 10.038297653198242, "learning_rate": 3.1967213114754097e-06, "loss": 0.6888828277587891, "step": 40 }, { "epoch": 0.1366120218579235, "grad_norm": 10.041961669921875, "learning_rate": 4.016393442622951e-06, "loss": 0.623580026626587, "step": 50 }, { "epoch": 0.16393442622950818, "grad_norm": 9.107032775878906, "learning_rate": 4.836065573770492e-06, "loss": 0.692430830001831, "step": 60 }, { "epoch": 0.1912568306010929, "grad_norm": 9.090417861938477, "learning_rate": 5.655737704918032e-06, "loss": 0.6110644817352295, "step": 70 }, { "epoch": 0.2185792349726776, "grad_norm": 7.914913177490234, "learning_rate": 6.4754098360655735e-06, "loss": 0.6268288612365722, "step": 80 }, { "epoch": 0.2459016393442623, "grad_norm": 8.18041706085205, "learning_rate": 7.2950819672131145e-06, "loss": 0.6859075546264648, "step": 90 }, { "epoch": 0.273224043715847, "grad_norm": 5.7157392501831055, "learning_rate": 8.114754098360657e-06, "loss": 0.7717578887939454, "step": 100 }, { "epoch": 0.3005464480874317, "grad_norm": 11.109936714172363, "learning_rate": 8.934426229508197e-06, "loss": 0.6476659774780273, "step": 110 }, { "epoch": 0.32786885245901637, "grad_norm": 10.369423866271973, "learning_rate": 9.754098360655738e-06, "loss": 0.6707509994506836, "step": 120 }, { "epoch": 0.3551912568306011, "grad_norm": 10.730874061584473, "learning_rate": 1.0573770491803279e-05, "loss": 0.6787842750549317, "step": 130 }, { "epoch": 0.3825136612021858, "grad_norm": 8.120079040527344, "learning_rate": 1.139344262295082e-05, "loss": 0.6765446186065673, "step": 140 }, { "epoch": 0.4098360655737705, "grad_norm": 14.429889678955078, "learning_rate": 1.221311475409836e-05, "loss": 0.623370361328125, "step": 150 }, { "epoch": 0.4371584699453552, "grad_norm": 9.193517684936523, "learning_rate": 1.3032786885245902e-05, "loss": 0.6362186908721924, "step": 160 }, { "epoch": 0.4644808743169399, "grad_norm": 8.7772855758667, "learning_rate": 1.3852459016393443e-05, "loss": 0.6641538619995118, "step": 170 }, { "epoch": 0.4918032786885246, "grad_norm": 5.2151780128479, "learning_rate": 1.4672131147540984e-05, "loss": 0.6653701782226562, "step": 180 }, { "epoch": 0.5191256830601093, "grad_norm": 17.197834014892578, "learning_rate": 1.5491803278688525e-05, "loss": 0.6582591533660889, "step": 190 }, { "epoch": 0.546448087431694, "grad_norm": 7.26473331451416, "learning_rate": 1.6311475409836068e-05, "loss": 0.765175724029541, "step": 200 }, { "epoch": 0.5737704918032787, "grad_norm": 5.47290563583374, "learning_rate": 1.7131147540983607e-05, "loss": 0.676247501373291, "step": 210 }, { "epoch": 0.6010928961748634, "grad_norm": 4.55684232711792, "learning_rate": 1.7950819672131146e-05, "loss": 0.6748649597167968, "step": 220 }, { "epoch": 0.6284153005464481, "grad_norm": 9.52070140838623, "learning_rate": 1.877049180327869e-05, "loss": 0.7023877143859864, "step": 230 }, { "epoch": 0.6557377049180327, "grad_norm": 10.813623428344727, "learning_rate": 1.959016393442623e-05, "loss": 0.5075790405273437, "step": 240 }, { "epoch": 0.6830601092896175, "grad_norm": 15.77322769165039, "learning_rate": 2.040983606557377e-05, "loss": 0.5947454929351806, "step": 250 }, { "epoch": 0.7103825136612022, "grad_norm": 14.141736030578613, "learning_rate": 2.122950819672131e-05, "loss": 0.7629669666290283, "step": 260 }, { "epoch": 0.7377049180327869, "grad_norm": 0.5761239528656006, "learning_rate": 2.2049180327868853e-05, "loss": 0.717600679397583, "step": 270 }, { "epoch": 0.7650273224043715, "grad_norm": 13.026739120483398, "learning_rate": 2.2868852459016393e-05, "loss": 1.5919431686401366, "step": 280 }, { "epoch": 0.7923497267759563, "grad_norm": 12.15133285522461, "learning_rate": 2.3688524590163936e-05, "loss": 0.6558570861816406, "step": 290 }, { "epoch": 0.819672131147541, "grad_norm": 20.35292625427246, "learning_rate": 2.4508196721311478e-05, "loss": 0.6991987705230713, "step": 300 }, { "epoch": 0.8469945355191257, "grad_norm": 9.59134292602539, "learning_rate": 2.5327868852459018e-05, "loss": 1.1239334106445313, "step": 310 }, { "epoch": 0.8743169398907104, "grad_norm": 4.265232086181641, "learning_rate": 2.6147540983606557e-05, "loss": 0.4866987705230713, "step": 320 }, { "epoch": 0.9016393442622951, "grad_norm": 9.87460708618164, "learning_rate": 2.69672131147541e-05, "loss": 0.7435293674468995, "step": 330 }, { "epoch": 0.9289617486338798, "grad_norm": 12.707701683044434, "learning_rate": 2.7786885245901642e-05, "loss": 0.9141542434692382, "step": 340 }, { "epoch": 0.9562841530054644, "grad_norm": 2.9138967990875244, "learning_rate": 2.860655737704918e-05, "loss": 0.5319347858428956, "step": 350 }, { "epoch": 0.9836065573770492, "grad_norm": 6.917504787445068, "learning_rate": 2.942622950819672e-05, "loss": 0.5438683509826661, "step": 360 }, { "epoch": 1.0, "eval_accuracy": 0.8492, "eval_confusion_matrix": [ [ 282, 0 ], [ 54, 22 ] ], "eval_f1": 0.449, "eval_loss": 0.41911664605140686, "eval_pr_auc": 0.6187, "eval_precision": 1.0, "eval_recall": 0.2895, "eval_roc_auc": 0.7485, "eval_runtime": 138.7598, "eval_samples_per_second": 2.58, "eval_steps_per_second": 0.649, "step": 366 } ], "logging_steps": 10, "max_steps": 3660, "num_input_tokens_seen": 0, "num_train_epochs": 10, "save_steps": 500, "stateful_callbacks": { "EarlyStoppingCallback": { "args": { "early_stopping_patience": 3, "early_stopping_threshold": 0.0 }, "attributes": { "early_stopping_patience_counter": 0 } }, "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 385194585047040.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }