{ "best_global_step": 5000, "best_metric": 0.02543150633573532, "best_model_checkpoint": "/cluster/scratch/heejdo/ArTS/Arts_t5_f1/checkpoint-5000", "epoch": 10.277492291880781, "eval_steps": 5000, "global_step": 20000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.2569373072970195, "grad_norm": 0.1743178367614746, "learning_rate": 4.9150306643368624e-05, "loss": 0.4766, "step": 500 }, { "epoch": 0.513874614594039, "grad_norm": 0.11835462599992752, "learning_rate": 4.8293760920957963e-05, "loss": 0.0343, "step": 1000 }, { "epoch": 0.7708119218910586, "grad_norm": 0.11048412322998047, "learning_rate": 4.74372151985473e-05, "loss": 0.0318, "step": 1500 }, { "epoch": 1.027749229188078, "grad_norm": 0.13272280991077423, "learning_rate": 4.658066947613664e-05, "loss": 0.0307, "step": 2000 }, { "epoch": 1.2846865364850977, "grad_norm": 0.09689372032880783, "learning_rate": 4.572412375372598e-05, "loss": 0.0291, "step": 2500 }, { "epoch": 1.541623843782117, "grad_norm": 0.12894241511821747, "learning_rate": 4.486757803131531e-05, "loss": 0.029, "step": 3000 }, { "epoch": 1.7985611510791366, "grad_norm": 0.08225054293870926, "learning_rate": 4.401103230890465e-05, "loss": 0.0284, "step": 3500 }, { "epoch": 2.055498458376156, "grad_norm": 0.08071951568126678, "learning_rate": 4.315448658649399e-05, "loss": 0.0281, "step": 4000 }, { "epoch": 2.3124357656731758, "grad_norm": 0.06852088868618011, "learning_rate": 4.2297940864083324e-05, "loss": 0.0266, "step": 4500 }, { "epoch": 2.5693730729701953, "grad_norm": 0.08975471556186676, "learning_rate": 4.144139514167266e-05, "loss": 0.0267, "step": 5000 }, { "epoch": 2.5693730729701953, "eval_loss": 0.02543150633573532, "eval_runtime": 183.8711, "eval_samples_per_second": 14.129, "eval_steps_per_second": 3.535, "step": 5000 }, { "epoch": 2.826310380267215, "grad_norm": 0.10155350714921951, "learning_rate": 4.0584849419262e-05, "loss": 0.0257, "step": 5500 }, { "epoch": 3.0832476875642345, "grad_norm": 0.10325942933559418, "learning_rate": 3.9728303696851334e-05, "loss": 0.0265, "step": 6000 }, { "epoch": 3.3401849948612536, "grad_norm": 0.14938808977603912, "learning_rate": 3.8871757974440674e-05, "loss": 0.025, "step": 6500 }, { "epoch": 3.597122302158273, "grad_norm": 0.1053280234336853, "learning_rate": 3.801521225203001e-05, "loss": 0.0243, "step": 7000 }, { "epoch": 3.854059609455293, "grad_norm": 0.07286415249109268, "learning_rate": 3.715866652961935e-05, "loss": 0.0241, "step": 7500 }, { "epoch": 4.110996916752312, "grad_norm": 0.08872249722480774, "learning_rate": 3.630212080720869e-05, "loss": 0.0232, "step": 8000 }, { "epoch": 4.367934224049332, "grad_norm": 0.11014755070209503, "learning_rate": 3.544557508479803e-05, "loss": 0.0218, "step": 8500 }, { "epoch": 4.6248715313463515, "grad_norm": 0.06434716284275055, "learning_rate": 3.458902936238736e-05, "loss": 0.0226, "step": 9000 }, { "epoch": 4.881808838643371, "grad_norm": 0.10045439004898071, "learning_rate": 3.37324836399767e-05, "loss": 0.0228, "step": 9500 }, { "epoch": 5.138746145940391, "grad_norm": 0.20274607837200165, "learning_rate": 3.287593791756604e-05, "loss": 0.0209, "step": 10000 }, { "epoch": 5.138746145940391, "eval_loss": 0.025791656225919724, "eval_runtime": 183.3324, "eval_samples_per_second": 14.171, "eval_steps_per_second": 3.545, "step": 10000 }, { "epoch": 5.39568345323741, "grad_norm": 0.20836800336837769, "learning_rate": 3.201939219515538e-05, "loss": 0.0201, "step": 10500 }, { "epoch": 5.65262076053443, "grad_norm": 0.10793159902095795, "learning_rate": 3.116284647274472e-05, "loss": 0.0198, "step": 11000 }, { "epoch": 5.909558067831449, "grad_norm": 0.13593679666519165, "learning_rate": 3.0306300750334055e-05, "loss": 0.0204, "step": 11500 }, { "epoch": 6.166495375128469, "grad_norm": 0.16172699630260468, "learning_rate": 2.9449755027923394e-05, "loss": 0.0188, "step": 12000 }, { "epoch": 6.423432682425489, "grad_norm": 0.09396054595708847, "learning_rate": 2.859320930551273e-05, "loss": 0.0175, "step": 12500 }, { "epoch": 6.680369989722507, "grad_norm": 0.11327569931745529, "learning_rate": 2.773666358310207e-05, "loss": 0.0182, "step": 13000 }, { "epoch": 6.937307297019527, "grad_norm": 0.09471254050731659, "learning_rate": 2.6880117860691405e-05, "loss": 0.0177, "step": 13500 }, { "epoch": 7.194244604316546, "grad_norm": 0.1715673953294754, "learning_rate": 2.6023572138280744e-05, "loss": 0.0158, "step": 14000 }, { "epoch": 7.451181911613566, "grad_norm": 0.13259157538414001, "learning_rate": 2.5167026415870083e-05, "loss": 0.0161, "step": 14500 }, { "epoch": 7.708119218910586, "grad_norm": 0.11879981309175491, "learning_rate": 2.431048069345942e-05, "loss": 0.0156, "step": 15000 }, { "epoch": 7.708119218910586, "eval_loss": 0.031164366751909256, "eval_runtime": 184.0426, "eval_samples_per_second": 14.116, "eval_steps_per_second": 3.532, "step": 15000 }, { "epoch": 7.965056526207605, "grad_norm": 0.19024206697940826, "learning_rate": 2.3453934971048754e-05, "loss": 0.0159, "step": 15500 }, { "epoch": 8.221993833504625, "grad_norm": 0.10826540738344193, "learning_rate": 2.2597389248638093e-05, "loss": 0.0139, "step": 16000 }, { "epoch": 8.478931140801645, "grad_norm": 0.08363316208124161, "learning_rate": 2.174084352622743e-05, "loss": 0.014, "step": 16500 }, { "epoch": 8.735868448098664, "grad_norm": 0.1762365847826004, "learning_rate": 2.088429780381677e-05, "loss": 0.0139, "step": 17000 }, { "epoch": 8.992805755395683, "grad_norm": 0.25474920868873596, "learning_rate": 2.0027752081406107e-05, "loss": 0.0139, "step": 17500 }, { "epoch": 9.249743062692703, "grad_norm": 0.402317613363266, "learning_rate": 1.9171206358995443e-05, "loss": 0.0121, "step": 18000 }, { "epoch": 9.506680369989722, "grad_norm": 0.32798829674720764, "learning_rate": 1.8314660636584782e-05, "loss": 0.012, "step": 18500 }, { "epoch": 9.763617677286742, "grad_norm": 0.2373279631137848, "learning_rate": 1.745811491417412e-05, "loss": 0.0125, "step": 19000 }, { "epoch": 10.020554984583761, "grad_norm": 0.13177204132080078, "learning_rate": 1.6601569191763457e-05, "loss": 0.0113, "step": 19500 }, { "epoch": 10.277492291880781, "grad_norm": 0.17192648351192474, "learning_rate": 1.5745023469352796e-05, "loss": 0.0107, "step": 20000 }, { "epoch": 10.277492291880781, "eval_loss": 0.0412333682179451, "eval_runtime": 182.8372, "eval_samples_per_second": 14.209, "eval_steps_per_second": 3.555, "step": 20000 } ], "logging_steps": 500, "max_steps": 29190, "num_input_tokens_seen": 0, "num_train_epochs": 15, "save_steps": 5000, "stateful_callbacks": { "EarlyStoppingCallback": { "args": { "early_stopping_patience": 3, "early_stopping_threshold": 0.0 }, "attributes": { "early_stopping_patience_counter": 3 } }, "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 1.7316048273408e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }