| { |
| "best_global_step": 10000, |
| "best_metric": 0.024344714358448982, |
| "best_model_checkpoint": "/cluster/project/sachan/heejin/RaDME/Arts_t5_f0/checkpoint-10000", |
| "epoch": 12.846865364850977, |
| "eval_steps": 5000, |
| "global_step": 25000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.2569373072970195, |
| "grad_norm": 0.17562313377857208, |
| "learning_rate": 4.9150306643368624e-05, |
| "loss": 0.5717, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.513874614594039, |
| "grad_norm": 0.10371646285057068, |
| "learning_rate": 4.8293760920957963e-05, |
| "loss": 0.0339, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.7708119218910586, |
| "grad_norm": 0.14652925729751587, |
| "learning_rate": 4.74372151985473e-05, |
| "loss": 0.0318, |
| "step": 1500 |
| }, |
| { |
| "epoch": 1.027749229188078, |
| "grad_norm": 0.13600969314575195, |
| "learning_rate": 4.658066947613664e-05, |
| "loss": 0.0311, |
| "step": 2000 |
| }, |
| { |
| "epoch": 1.2846865364850977, |
| "grad_norm": 0.0706079825758934, |
| "learning_rate": 4.572412375372598e-05, |
| "loss": 0.0295, |
| "step": 2500 |
| }, |
| { |
| "epoch": 1.541623843782117, |
| "grad_norm": 1.7536998987197876, |
| "learning_rate": 4.486757803131531e-05, |
| "loss": 0.0315, |
| "step": 3000 |
| }, |
| { |
| "epoch": 1.7985611510791366, |
| "grad_norm": 0.10474766790866852, |
| "learning_rate": 4.401103230890465e-05, |
| "loss": 0.029, |
| "step": 3500 |
| }, |
| { |
| "epoch": 2.055498458376156, |
| "grad_norm": 0.549686849117279, |
| "learning_rate": 4.315448658649399e-05, |
| "loss": 0.0284, |
| "step": 4000 |
| }, |
| { |
| "epoch": 2.3124357656731758, |
| "grad_norm": 0.09790255129337311, |
| "learning_rate": 4.2297940864083324e-05, |
| "loss": 0.0283, |
| "step": 4500 |
| }, |
| { |
| "epoch": 2.5693730729701953, |
| "grad_norm": 0.09890051931142807, |
| "learning_rate": 4.144139514167266e-05, |
| "loss": 0.0269, |
| "step": 5000 |
| }, |
| { |
| "epoch": 2.5693730729701953, |
| "eval_loss": 0.02565159648656845, |
| "eval_runtime": 178.3077, |
| "eval_samples_per_second": 14.542, |
| "eval_steps_per_second": 3.64, |
| "step": 5000 |
| }, |
| { |
| "epoch": 2.826310380267215, |
| "grad_norm": 0.10465002059936523, |
| "learning_rate": 4.0584849419262e-05, |
| "loss": 0.0266, |
| "step": 5500 |
| }, |
| { |
| "epoch": 3.0832476875642345, |
| "grad_norm": 0.10655808448791504, |
| "learning_rate": 3.9728303696851334e-05, |
| "loss": 0.0266, |
| "step": 6000 |
| }, |
| { |
| "epoch": 3.3401849948612536, |
| "grad_norm": 0.09030739963054657, |
| "learning_rate": 3.8871757974440674e-05, |
| "loss": 0.0257, |
| "step": 6500 |
| }, |
| { |
| "epoch": 3.597122302158273, |
| "grad_norm": 0.077540323138237, |
| "learning_rate": 3.801521225203001e-05, |
| "loss": 0.0247, |
| "step": 7000 |
| }, |
| { |
| "epoch": 3.854059609455293, |
| "grad_norm": 0.17392417788505554, |
| "learning_rate": 3.715866652961935e-05, |
| "loss": 0.0256, |
| "step": 7500 |
| }, |
| { |
| "epoch": 4.110996916752312, |
| "grad_norm": 0.3808821141719818, |
| "learning_rate": 3.630212080720869e-05, |
| "loss": 0.0245, |
| "step": 8000 |
| }, |
| { |
| "epoch": 4.367934224049332, |
| "grad_norm": 0.07896172255277634, |
| "learning_rate": 3.544557508479803e-05, |
| "loss": 0.0239, |
| "step": 8500 |
| }, |
| { |
| "epoch": 4.6248715313463515, |
| "grad_norm": 0.08702714741230011, |
| "learning_rate": 3.458902936238736e-05, |
| "loss": 0.0238, |
| "step": 9000 |
| }, |
| { |
| "epoch": 4.881808838643371, |
| "grad_norm": 0.10166553407907486, |
| "learning_rate": 3.37324836399767e-05, |
| "loss": 0.0238, |
| "step": 9500 |
| }, |
| { |
| "epoch": 5.138746145940391, |
| "grad_norm": 0.11367424577474594, |
| "learning_rate": 3.287593791756604e-05, |
| "loss": 0.0227, |
| "step": 10000 |
| }, |
| { |
| "epoch": 5.138746145940391, |
| "eval_loss": 0.024344714358448982, |
| "eval_runtime": 178.7234, |
| "eval_samples_per_second": 14.508, |
| "eval_steps_per_second": 3.631, |
| "step": 10000 |
| }, |
| { |
| "epoch": 5.39568345323741, |
| "grad_norm": 0.06953849643468857, |
| "learning_rate": 3.201939219515538e-05, |
| "loss": 0.0222, |
| "step": 10500 |
| }, |
| { |
| "epoch": 5.65262076053443, |
| "grad_norm": 0.14536646008491516, |
| "learning_rate": 3.116284647274472e-05, |
| "loss": 0.0217, |
| "step": 11000 |
| }, |
| { |
| "epoch": 5.909558067831449, |
| "grad_norm": 0.1506756693124771, |
| "learning_rate": 3.0306300750334055e-05, |
| "loss": 0.0221, |
| "step": 11500 |
| }, |
| { |
| "epoch": 6.166495375128469, |
| "grad_norm": 0.07476528733968735, |
| "learning_rate": 2.9449755027923394e-05, |
| "loss": 0.0217, |
| "step": 12000 |
| }, |
| { |
| "epoch": 6.423432682425489, |
| "grad_norm": 0.11400212347507477, |
| "learning_rate": 2.859320930551273e-05, |
| "loss": 0.0208, |
| "step": 12500 |
| }, |
| { |
| "epoch": 6.680369989722507, |
| "grad_norm": 0.07216424494981766, |
| "learning_rate": 2.773666358310207e-05, |
| "loss": 0.0204, |
| "step": 13000 |
| }, |
| { |
| "epoch": 6.937307297019527, |
| "grad_norm": 0.12992289662361145, |
| "learning_rate": 2.6880117860691405e-05, |
| "loss": 0.02, |
| "step": 13500 |
| }, |
| { |
| "epoch": 7.194244604316546, |
| "grad_norm": 0.06588713079690933, |
| "learning_rate": 2.6023572138280744e-05, |
| "loss": 0.0191, |
| "step": 14000 |
| }, |
| { |
| "epoch": 7.451181911613566, |
| "grad_norm": 0.1124558374285698, |
| "learning_rate": 2.5167026415870083e-05, |
| "loss": 0.0186, |
| "step": 14500 |
| }, |
| { |
| "epoch": 7.708119218910586, |
| "grad_norm": 0.11662375181913376, |
| "learning_rate": 2.431048069345942e-05, |
| "loss": 0.0188, |
| "step": 15000 |
| }, |
| { |
| "epoch": 7.708119218910586, |
| "eval_loss": 0.027120616286993027, |
| "eval_runtime": 179.5611, |
| "eval_samples_per_second": 14.441, |
| "eval_steps_per_second": 3.614, |
| "step": 15000 |
| }, |
| { |
| "epoch": 7.965056526207605, |
| "grad_norm": 0.05247975140810013, |
| "learning_rate": 2.3453934971048754e-05, |
| "loss": 0.0185, |
| "step": 15500 |
| }, |
| { |
| "epoch": 8.221993833504625, |
| "grad_norm": 0.1325894296169281, |
| "learning_rate": 2.2597389248638093e-05, |
| "loss": 0.0167, |
| "step": 16000 |
| }, |
| { |
| "epoch": 8.478931140801645, |
| "grad_norm": 0.1109691709280014, |
| "learning_rate": 2.174084352622743e-05, |
| "loss": 0.0171, |
| "step": 16500 |
| }, |
| { |
| "epoch": 8.735868448098664, |
| "grad_norm": 0.1451205462217331, |
| "learning_rate": 2.088429780381677e-05, |
| "loss": 0.0167, |
| "step": 17000 |
| }, |
| { |
| "epoch": 8.992805755395683, |
| "grad_norm": 0.18753716349601746, |
| "learning_rate": 2.0027752081406107e-05, |
| "loss": 0.0176, |
| "step": 17500 |
| }, |
| { |
| "epoch": 9.249743062692703, |
| "grad_norm": 0.3178722858428955, |
| "learning_rate": 1.9171206358995443e-05, |
| "loss": 0.0151, |
| "step": 18000 |
| }, |
| { |
| "epoch": 9.506680369989722, |
| "grad_norm": 0.23986810445785522, |
| "learning_rate": 1.8314660636584782e-05, |
| "loss": 0.0156, |
| "step": 18500 |
| }, |
| { |
| "epoch": 9.763617677286742, |
| "grad_norm": 0.1744610071182251, |
| "learning_rate": 1.745811491417412e-05, |
| "loss": 0.0154, |
| "step": 19000 |
| }, |
| { |
| "epoch": 10.020554984583761, |
| "grad_norm": 0.23001733422279358, |
| "learning_rate": 1.6601569191763457e-05, |
| "loss": 0.0155, |
| "step": 19500 |
| }, |
| { |
| "epoch": 10.277492291880781, |
| "grad_norm": 0.17580315470695496, |
| "learning_rate": 1.5745023469352796e-05, |
| "loss": 0.0136, |
| "step": 20000 |
| }, |
| { |
| "epoch": 10.277492291880781, |
| "eval_loss": 0.03201943263411522, |
| "eval_runtime": 184.1169, |
| "eval_samples_per_second": 14.083, |
| "eval_steps_per_second": 3.525, |
| "step": 20000 |
| }, |
| { |
| "epoch": 10.5344295991778, |
| "grad_norm": 0.137004092335701, |
| "learning_rate": 1.4888477746942134e-05, |
| "loss": 0.0134, |
| "step": 20500 |
| }, |
| { |
| "epoch": 10.79136690647482, |
| "grad_norm": 0.209578737616539, |
| "learning_rate": 1.403193202453147e-05, |
| "loss": 0.0138, |
| "step": 21000 |
| }, |
| { |
| "epoch": 11.04830421377184, |
| "grad_norm": 0.17923501133918762, |
| "learning_rate": 1.3175386302120807e-05, |
| "loss": 0.0135, |
| "step": 21500 |
| }, |
| { |
| "epoch": 11.30524152106886, |
| "grad_norm": 0.23782482743263245, |
| "learning_rate": 1.2318840579710146e-05, |
| "loss": 0.0125, |
| "step": 22000 |
| }, |
| { |
| "epoch": 11.562178828365878, |
| "grad_norm": 0.1997631937265396, |
| "learning_rate": 1.1462294857299482e-05, |
| "loss": 0.0123, |
| "step": 22500 |
| }, |
| { |
| "epoch": 11.819116135662899, |
| "grad_norm": 0.185743510723114, |
| "learning_rate": 1.0605749134888821e-05, |
| "loss": 0.0122, |
| "step": 23000 |
| }, |
| { |
| "epoch": 12.076053442959918, |
| "grad_norm": 0.21140512824058533, |
| "learning_rate": 9.749203412478158e-06, |
| "loss": 0.012, |
| "step": 23500 |
| }, |
| { |
| "epoch": 12.332990750256938, |
| "grad_norm": 0.1282157599925995, |
| "learning_rate": 8.892657690067496e-06, |
| "loss": 0.0114, |
| "step": 24000 |
| }, |
| { |
| "epoch": 12.589928057553957, |
| "grad_norm": 0.23103195428848267, |
| "learning_rate": 8.036111967656835e-06, |
| "loss": 0.0111, |
| "step": 24500 |
| }, |
| { |
| "epoch": 12.846865364850977, |
| "grad_norm": 0.14510518312454224, |
| "learning_rate": 7.1795662452461724e-06, |
| "loss": 0.0112, |
| "step": 25000 |
| }, |
| { |
| "epoch": 12.846865364850977, |
| "eval_loss": 0.038895536214113235, |
| "eval_runtime": 178.5696, |
| "eval_samples_per_second": 14.521, |
| "eval_steps_per_second": 3.634, |
| "step": 25000 |
| } |
| ], |
| "logging_steps": 500, |
| "max_steps": 29190, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 15, |
| "save_steps": 5000, |
| "stateful_callbacks": { |
| "EarlyStoppingCallback": { |
| "args": { |
| "early_stopping_patience": 3, |
| "early_stopping_threshold": 0.0 |
| }, |
| "attributes": { |
| "early_stopping_patience_counter": 3 |
| } |
| }, |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 2.165047296e+17, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|