| { |
| "best_global_step": 5000, |
| "best_metric": 0.02543150633573532, |
| "best_model_checkpoint": "/cluster/scratch/heejdo/ArTS/Arts_t5_f1/checkpoint-5000", |
| "epoch": 10.277492291880781, |
| "eval_steps": 5000, |
| "global_step": 20000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.2569373072970195, |
| "grad_norm": 0.1743178367614746, |
| "learning_rate": 4.9150306643368624e-05, |
| "loss": 0.4766, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.513874614594039, |
| "grad_norm": 0.11835462599992752, |
| "learning_rate": 4.8293760920957963e-05, |
| "loss": 0.0343, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.7708119218910586, |
| "grad_norm": 0.11048412322998047, |
| "learning_rate": 4.74372151985473e-05, |
| "loss": 0.0318, |
| "step": 1500 |
| }, |
| { |
| "epoch": 1.027749229188078, |
| "grad_norm": 0.13272280991077423, |
| "learning_rate": 4.658066947613664e-05, |
| "loss": 0.0307, |
| "step": 2000 |
| }, |
| { |
| "epoch": 1.2846865364850977, |
| "grad_norm": 0.09689372032880783, |
| "learning_rate": 4.572412375372598e-05, |
| "loss": 0.0291, |
| "step": 2500 |
| }, |
| { |
| "epoch": 1.541623843782117, |
| "grad_norm": 0.12894241511821747, |
| "learning_rate": 4.486757803131531e-05, |
| "loss": 0.029, |
| "step": 3000 |
| }, |
| { |
| "epoch": 1.7985611510791366, |
| "grad_norm": 0.08225054293870926, |
| "learning_rate": 4.401103230890465e-05, |
| "loss": 0.0284, |
| "step": 3500 |
| }, |
| { |
| "epoch": 2.055498458376156, |
| "grad_norm": 0.08071951568126678, |
| "learning_rate": 4.315448658649399e-05, |
| "loss": 0.0281, |
| "step": 4000 |
| }, |
| { |
| "epoch": 2.3124357656731758, |
| "grad_norm": 0.06852088868618011, |
| "learning_rate": 4.2297940864083324e-05, |
| "loss": 0.0266, |
| "step": 4500 |
| }, |
| { |
| "epoch": 2.5693730729701953, |
| "grad_norm": 0.08975471556186676, |
| "learning_rate": 4.144139514167266e-05, |
| "loss": 0.0267, |
| "step": 5000 |
| }, |
| { |
| "epoch": 2.5693730729701953, |
| "eval_loss": 0.02543150633573532, |
| "eval_runtime": 183.8711, |
| "eval_samples_per_second": 14.129, |
| "eval_steps_per_second": 3.535, |
| "step": 5000 |
| }, |
| { |
| "epoch": 2.826310380267215, |
| "grad_norm": 0.10155350714921951, |
| "learning_rate": 4.0584849419262e-05, |
| "loss": 0.0257, |
| "step": 5500 |
| }, |
| { |
| "epoch": 3.0832476875642345, |
| "grad_norm": 0.10325942933559418, |
| "learning_rate": 3.9728303696851334e-05, |
| "loss": 0.0265, |
| "step": 6000 |
| }, |
| { |
| "epoch": 3.3401849948612536, |
| "grad_norm": 0.14938808977603912, |
| "learning_rate": 3.8871757974440674e-05, |
| "loss": 0.025, |
| "step": 6500 |
| }, |
| { |
| "epoch": 3.597122302158273, |
| "grad_norm": 0.1053280234336853, |
| "learning_rate": 3.801521225203001e-05, |
| "loss": 0.0243, |
| "step": 7000 |
| }, |
| { |
| "epoch": 3.854059609455293, |
| "grad_norm": 0.07286415249109268, |
| "learning_rate": 3.715866652961935e-05, |
| "loss": 0.0241, |
| "step": 7500 |
| }, |
| { |
| "epoch": 4.110996916752312, |
| "grad_norm": 0.08872249722480774, |
| "learning_rate": 3.630212080720869e-05, |
| "loss": 0.0232, |
| "step": 8000 |
| }, |
| { |
| "epoch": 4.367934224049332, |
| "grad_norm": 0.11014755070209503, |
| "learning_rate": 3.544557508479803e-05, |
| "loss": 0.0218, |
| "step": 8500 |
| }, |
| { |
| "epoch": 4.6248715313463515, |
| "grad_norm": 0.06434716284275055, |
| "learning_rate": 3.458902936238736e-05, |
| "loss": 0.0226, |
| "step": 9000 |
| }, |
| { |
| "epoch": 4.881808838643371, |
| "grad_norm": 0.10045439004898071, |
| "learning_rate": 3.37324836399767e-05, |
| "loss": 0.0228, |
| "step": 9500 |
| }, |
| { |
| "epoch": 5.138746145940391, |
| "grad_norm": 0.20274607837200165, |
| "learning_rate": 3.287593791756604e-05, |
| "loss": 0.0209, |
| "step": 10000 |
| }, |
| { |
| "epoch": 5.138746145940391, |
| "eval_loss": 0.025791656225919724, |
| "eval_runtime": 183.3324, |
| "eval_samples_per_second": 14.171, |
| "eval_steps_per_second": 3.545, |
| "step": 10000 |
| }, |
| { |
| "epoch": 5.39568345323741, |
| "grad_norm": 0.20836800336837769, |
| "learning_rate": 3.201939219515538e-05, |
| "loss": 0.0201, |
| "step": 10500 |
| }, |
| { |
| "epoch": 5.65262076053443, |
| "grad_norm": 0.10793159902095795, |
| "learning_rate": 3.116284647274472e-05, |
| "loss": 0.0198, |
| "step": 11000 |
| }, |
| { |
| "epoch": 5.909558067831449, |
| "grad_norm": 0.13593679666519165, |
| "learning_rate": 3.0306300750334055e-05, |
| "loss": 0.0204, |
| "step": 11500 |
| }, |
| { |
| "epoch": 6.166495375128469, |
| "grad_norm": 0.16172699630260468, |
| "learning_rate": 2.9449755027923394e-05, |
| "loss": 0.0188, |
| "step": 12000 |
| }, |
| { |
| "epoch": 6.423432682425489, |
| "grad_norm": 0.09396054595708847, |
| "learning_rate": 2.859320930551273e-05, |
| "loss": 0.0175, |
| "step": 12500 |
| }, |
| { |
| "epoch": 6.680369989722507, |
| "grad_norm": 0.11327569931745529, |
| "learning_rate": 2.773666358310207e-05, |
| "loss": 0.0182, |
| "step": 13000 |
| }, |
| { |
| "epoch": 6.937307297019527, |
| "grad_norm": 0.09471254050731659, |
| "learning_rate": 2.6880117860691405e-05, |
| "loss": 0.0177, |
| "step": 13500 |
| }, |
| { |
| "epoch": 7.194244604316546, |
| "grad_norm": 0.1715673953294754, |
| "learning_rate": 2.6023572138280744e-05, |
| "loss": 0.0158, |
| "step": 14000 |
| }, |
| { |
| "epoch": 7.451181911613566, |
| "grad_norm": 0.13259157538414001, |
| "learning_rate": 2.5167026415870083e-05, |
| "loss": 0.0161, |
| "step": 14500 |
| }, |
| { |
| "epoch": 7.708119218910586, |
| "grad_norm": 0.11879981309175491, |
| "learning_rate": 2.431048069345942e-05, |
| "loss": 0.0156, |
| "step": 15000 |
| }, |
| { |
| "epoch": 7.708119218910586, |
| "eval_loss": 0.031164366751909256, |
| "eval_runtime": 184.0426, |
| "eval_samples_per_second": 14.116, |
| "eval_steps_per_second": 3.532, |
| "step": 15000 |
| }, |
| { |
| "epoch": 7.965056526207605, |
| "grad_norm": 0.19024206697940826, |
| "learning_rate": 2.3453934971048754e-05, |
| "loss": 0.0159, |
| "step": 15500 |
| }, |
| { |
| "epoch": 8.221993833504625, |
| "grad_norm": 0.10826540738344193, |
| "learning_rate": 2.2597389248638093e-05, |
| "loss": 0.0139, |
| "step": 16000 |
| }, |
| { |
| "epoch": 8.478931140801645, |
| "grad_norm": 0.08363316208124161, |
| "learning_rate": 2.174084352622743e-05, |
| "loss": 0.014, |
| "step": 16500 |
| }, |
| { |
| "epoch": 8.735868448098664, |
| "grad_norm": 0.1762365847826004, |
| "learning_rate": 2.088429780381677e-05, |
| "loss": 0.0139, |
| "step": 17000 |
| }, |
| { |
| "epoch": 8.992805755395683, |
| "grad_norm": 0.25474920868873596, |
| "learning_rate": 2.0027752081406107e-05, |
| "loss": 0.0139, |
| "step": 17500 |
| }, |
| { |
| "epoch": 9.249743062692703, |
| "grad_norm": 0.402317613363266, |
| "learning_rate": 1.9171206358995443e-05, |
| "loss": 0.0121, |
| "step": 18000 |
| }, |
| { |
| "epoch": 9.506680369989722, |
| "grad_norm": 0.32798829674720764, |
| "learning_rate": 1.8314660636584782e-05, |
| "loss": 0.012, |
| "step": 18500 |
| }, |
| { |
| "epoch": 9.763617677286742, |
| "grad_norm": 0.2373279631137848, |
| "learning_rate": 1.745811491417412e-05, |
| "loss": 0.0125, |
| "step": 19000 |
| }, |
| { |
| "epoch": 10.020554984583761, |
| "grad_norm": 0.13177204132080078, |
| "learning_rate": 1.6601569191763457e-05, |
| "loss": 0.0113, |
| "step": 19500 |
| }, |
| { |
| "epoch": 10.277492291880781, |
| "grad_norm": 0.17192648351192474, |
| "learning_rate": 1.5745023469352796e-05, |
| "loss": 0.0107, |
| "step": 20000 |
| }, |
| { |
| "epoch": 10.277492291880781, |
| "eval_loss": 0.0412333682179451, |
| "eval_runtime": 182.8372, |
| "eval_samples_per_second": 14.209, |
| "eval_steps_per_second": 3.555, |
| "step": 20000 |
| } |
| ], |
| "logging_steps": 500, |
| "max_steps": 29190, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 15, |
| "save_steps": 5000, |
| "stateful_callbacks": { |
| "EarlyStoppingCallback": { |
| "args": { |
| "early_stopping_patience": 3, |
| "early_stopping_threshold": 0.0 |
| }, |
| "attributes": { |
| "early_stopping_patience_counter": 3 |
| } |
| }, |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1.7316048273408e+17, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|