{ "best_global_step": 10000, "best_metric": 0.026011742651462555, "best_model_checkpoint": "/cluster/scratch/heejdo/ArTS/Arts_t5_f2/checkpoint-10000", "epoch": 12.840267077555213, "eval_steps": 5000, "global_step": 25000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.25680534155110424, "grad_norm": 0.4051225185394287, "learning_rate": 4.9150743099787686e-05, "loss": 0.564, "step": 500 }, { "epoch": 0.5136106831022085, "grad_norm": 0.2164618968963623, "learning_rate": 4.829463735360592e-05, "loss": 0.0337, "step": 1000 }, { "epoch": 0.7704160246533128, "grad_norm": 0.14942540228366852, "learning_rate": 4.743853160742415e-05, "loss": 0.0316, "step": 1500 }, { "epoch": 1.027221366204417, "grad_norm": 0.1534731388092041, "learning_rate": 4.6582425861242384e-05, "loss": 0.0303, "step": 2000 }, { "epoch": 1.2840267077555212, "grad_norm": 0.11905260384082794, "learning_rate": 4.5726320115060614e-05, "loss": 0.0296, "step": 2500 }, { "epoch": 1.5408320493066254, "grad_norm": 0.13345952332019806, "learning_rate": 4.4870214368878845e-05, "loss": 0.0287, "step": 3000 }, { "epoch": 1.7976373908577299, "grad_norm": 0.10080758482217789, "learning_rate": 4.4014108622697075e-05, "loss": 0.0278, "step": 3500 }, { "epoch": 2.054442732408834, "grad_norm": 0.10812211036682129, "learning_rate": 4.3158002876515306e-05, "loss": 0.0273, "step": 4000 }, { "epoch": 2.3112480739599386, "grad_norm": 0.11209239810705185, "learning_rate": 4.230189713033354e-05, "loss": 0.0264, "step": 4500 }, { "epoch": 2.5680534155110424, "grad_norm": 0.06536886841058731, "learning_rate": 4.1445791384151774e-05, "loss": 0.0263, "step": 5000 }, { "epoch": 2.5680534155110424, "eval_loss": 0.02743290737271309, "eval_runtime": 181.2622, "eval_samples_per_second": 14.316, "eval_steps_per_second": 3.58, "step": 5000 }, { "epoch": 2.824858757062147, "grad_norm": 0.07943771034479141, "learning_rate": 4.0589685637970004e-05, "loss": 0.0262, "step": 5500 }, { "epoch": 3.0816640986132513, "grad_norm": 0.14247220754623413, "learning_rate": 3.9733579891788235e-05, "loss": 0.0252, "step": 6000 }, { "epoch": 3.3384694401643555, "grad_norm": 0.06139238923788071, "learning_rate": 3.8877474145606465e-05, "loss": 0.0244, "step": 6500 }, { "epoch": 3.5952747817154598, "grad_norm": 0.09731902182102203, "learning_rate": 3.80213683994247e-05, "loss": 0.0244, "step": 7000 }, { "epoch": 3.852080123266564, "grad_norm": 0.07361181080341339, "learning_rate": 3.716526265324293e-05, "loss": 0.0242, "step": 7500 }, { "epoch": 4.108885464817668, "grad_norm": 0.05958514288067818, "learning_rate": 3.630915690706116e-05, "loss": 0.0233, "step": 8000 }, { "epoch": 4.3656908063687725, "grad_norm": 0.1113390326499939, "learning_rate": 3.5453051160879394e-05, "loss": 0.0224, "step": 8500 }, { "epoch": 4.622496147919877, "grad_norm": 0.09913190454244614, "learning_rate": 3.4596945414697624e-05, "loss": 0.0227, "step": 9000 }, { "epoch": 4.879301489470981, "grad_norm": 0.13124661147594452, "learning_rate": 3.3740839668515855e-05, "loss": 0.0221, "step": 9500 }, { "epoch": 5.136106831022086, "grad_norm": 0.09466934204101562, "learning_rate": 3.2884733922334085e-05, "loss": 0.0215, "step": 10000 }, { "epoch": 5.136106831022086, "eval_loss": 0.026011742651462555, "eval_runtime": 181.145, "eval_samples_per_second": 14.326, "eval_steps_per_second": 3.583, "step": 10000 }, { "epoch": 5.3929121725731894, "grad_norm": 0.13241207599639893, "learning_rate": 3.202862817615232e-05, "loss": 0.0203, "step": 10500 }, { "epoch": 5.649717514124294, "grad_norm": 0.10956753045320511, "learning_rate": 3.117252242997055e-05, "loss": 0.0205, "step": 11000 }, { "epoch": 5.906522855675398, "grad_norm": 0.1889301985502243, "learning_rate": 3.0316416683788784e-05, "loss": 0.0206, "step": 11500 }, { "epoch": 6.163328197226503, "grad_norm": 0.12303808331489563, "learning_rate": 2.9460310937607018e-05, "loss": 0.019, "step": 12000 }, { "epoch": 6.420133538777606, "grad_norm": 0.13128505647182465, "learning_rate": 2.8604205191425248e-05, "loss": 0.0189, "step": 12500 }, { "epoch": 6.676938880328711, "grad_norm": 0.23012486100196838, "learning_rate": 2.7748099445243475e-05, "loss": 0.0176, "step": 13000 }, { "epoch": 6.933744221879815, "grad_norm": 0.16152064502239227, "learning_rate": 2.6891993699061706e-05, "loss": 0.0186, "step": 13500 }, { "epoch": 7.1905495634309196, "grad_norm": 0.11679325997829437, "learning_rate": 2.603588795287994e-05, "loss": 0.0176, "step": 14000 }, { "epoch": 7.447354904982023, "grad_norm": 0.21249671280384064, "learning_rate": 2.517978220669817e-05, "loss": 0.0166, "step": 14500 }, { "epoch": 7.704160246533128, "grad_norm": 0.18364188075065613, "learning_rate": 2.4323676460516404e-05, "loss": 0.0166, "step": 15000 }, { "epoch": 7.704160246533128, "eval_loss": 0.03057297319173813, "eval_runtime": 185.7853, "eval_samples_per_second": 13.968, "eval_steps_per_second": 3.493, "step": 15000 }, { "epoch": 7.960965588084232, "grad_norm": 0.2111167013645172, "learning_rate": 2.3467570714334635e-05, "loss": 0.0167, "step": 15500 }, { "epoch": 8.217770929635336, "grad_norm": 0.3737495243549347, "learning_rate": 2.261146496815287e-05, "loss": 0.0147, "step": 16000 }, { "epoch": 8.474576271186441, "grad_norm": 0.2299918681383133, "learning_rate": 2.17553592219711e-05, "loss": 0.0145, "step": 16500 }, { "epoch": 8.731381612737545, "grad_norm": 0.151510551571846, "learning_rate": 2.089925347578933e-05, "loss": 0.0143, "step": 17000 }, { "epoch": 8.988186954288649, "grad_norm": 0.16585668921470642, "learning_rate": 2.004314772960756e-05, "loss": 0.015, "step": 17500 }, { "epoch": 9.244992295839754, "grad_norm": 0.14493027329444885, "learning_rate": 1.9187041983425794e-05, "loss": 0.0133, "step": 18000 }, { "epoch": 9.501797637390858, "grad_norm": 0.30722230672836304, "learning_rate": 1.8330936237244024e-05, "loss": 0.0129, "step": 18500 }, { "epoch": 9.758602978941962, "grad_norm": 0.09673864394426346, "learning_rate": 1.747483049106226e-05, "loss": 0.0127, "step": 19000 }, { "epoch": 10.015408320493066, "grad_norm": 0.2731100916862488, "learning_rate": 1.661872474488049e-05, "loss": 0.013, "step": 19500 }, { "epoch": 10.272213662044171, "grad_norm": 0.11305446922779083, "learning_rate": 1.576261899869872e-05, "loss": 0.0112, "step": 20000 }, { "epoch": 10.272213662044171, "eval_loss": 0.042304039001464844, "eval_runtime": 182.015, "eval_samples_per_second": 14.257, "eval_steps_per_second": 3.566, "step": 20000 }, { "epoch": 10.529019003595275, "grad_norm": 0.30115944147109985, "learning_rate": 1.4906513252516952e-05, "loss": 0.0118, "step": 20500 }, { "epoch": 10.785824345146379, "grad_norm": 0.22803886234760284, "learning_rate": 1.4050407506335184e-05, "loss": 0.011, "step": 21000 }, { "epoch": 11.042629686697483, "grad_norm": 0.08788817375898361, "learning_rate": 1.3194301760153416e-05, "loss": 0.0114, "step": 21500 }, { "epoch": 11.299435028248588, "grad_norm": 0.47961196303367615, "learning_rate": 1.2338196013971646e-05, "loss": 0.0102, "step": 22000 }, { "epoch": 11.556240369799692, "grad_norm": 0.3627168834209442, "learning_rate": 1.1482090267789877e-05, "loss": 0.01, "step": 22500 }, { "epoch": 11.813045711350796, "grad_norm": 0.22580872476100922, "learning_rate": 1.062598452160811e-05, "loss": 0.0103, "step": 23000 }, { "epoch": 12.0698510529019, "grad_norm": 0.26205044984817505, "learning_rate": 9.769878775426341e-06, "loss": 0.0098, "step": 23500 }, { "epoch": 12.326656394453005, "grad_norm": 0.16040152311325073, "learning_rate": 8.913773029244572e-06, "loss": 0.0093, "step": 24000 }, { "epoch": 12.583461736004109, "grad_norm": 0.1721157431602478, "learning_rate": 8.057667283062804e-06, "loss": 0.0093, "step": 24500 }, { "epoch": 12.840267077555213, "grad_norm": 0.583092212677002, "learning_rate": 7.201561536881036e-06, "loss": 0.0088, "step": 25000 }, { "epoch": 12.840267077555213, "eval_loss": 0.05068250373005867, "eval_runtime": 180.1951, "eval_samples_per_second": 14.401, "eval_steps_per_second": 3.602, "step": 25000 } ], "logging_steps": 500, "max_steps": 29205, "num_input_tokens_seen": 0, "num_train_epochs": 15, "save_steps": 5000, "stateful_callbacks": { "EarlyStoppingCallback": { "args": { "early_stopping_patience": 3, "early_stopping_threshold": 0.0 }, "attributes": { "early_stopping_patience_counter": 3 } }, "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 2.16426787897344e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }