| { |
| "best_global_step": 10000, |
| "best_metric": 0.026011742651462555, |
| "best_model_checkpoint": "/cluster/scratch/heejdo/ArTS/Arts_t5_f2/checkpoint-10000", |
| "epoch": 12.840267077555213, |
| "eval_steps": 5000, |
| "global_step": 25000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.25680534155110424, |
| "grad_norm": 0.4051225185394287, |
| "learning_rate": 4.9150743099787686e-05, |
| "loss": 0.564, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.5136106831022085, |
| "grad_norm": 0.2164618968963623, |
| "learning_rate": 4.829463735360592e-05, |
| "loss": 0.0337, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.7704160246533128, |
| "grad_norm": 0.14942540228366852, |
| "learning_rate": 4.743853160742415e-05, |
| "loss": 0.0316, |
| "step": 1500 |
| }, |
| { |
| "epoch": 1.027221366204417, |
| "grad_norm": 0.1534731388092041, |
| "learning_rate": 4.6582425861242384e-05, |
| "loss": 0.0303, |
| "step": 2000 |
| }, |
| { |
| "epoch": 1.2840267077555212, |
| "grad_norm": 0.11905260384082794, |
| "learning_rate": 4.5726320115060614e-05, |
| "loss": 0.0296, |
| "step": 2500 |
| }, |
| { |
| "epoch": 1.5408320493066254, |
| "grad_norm": 0.13345952332019806, |
| "learning_rate": 4.4870214368878845e-05, |
| "loss": 0.0287, |
| "step": 3000 |
| }, |
| { |
| "epoch": 1.7976373908577299, |
| "grad_norm": 0.10080758482217789, |
| "learning_rate": 4.4014108622697075e-05, |
| "loss": 0.0278, |
| "step": 3500 |
| }, |
| { |
| "epoch": 2.054442732408834, |
| "grad_norm": 0.10812211036682129, |
| "learning_rate": 4.3158002876515306e-05, |
| "loss": 0.0273, |
| "step": 4000 |
| }, |
| { |
| "epoch": 2.3112480739599386, |
| "grad_norm": 0.11209239810705185, |
| "learning_rate": 4.230189713033354e-05, |
| "loss": 0.0264, |
| "step": 4500 |
| }, |
| { |
| "epoch": 2.5680534155110424, |
| "grad_norm": 0.06536886841058731, |
| "learning_rate": 4.1445791384151774e-05, |
| "loss": 0.0263, |
| "step": 5000 |
| }, |
| { |
| "epoch": 2.5680534155110424, |
| "eval_loss": 0.02743290737271309, |
| "eval_runtime": 181.2622, |
| "eval_samples_per_second": 14.316, |
| "eval_steps_per_second": 3.58, |
| "step": 5000 |
| }, |
| { |
| "epoch": 2.824858757062147, |
| "grad_norm": 0.07943771034479141, |
| "learning_rate": 4.0589685637970004e-05, |
| "loss": 0.0262, |
| "step": 5500 |
| }, |
| { |
| "epoch": 3.0816640986132513, |
| "grad_norm": 0.14247220754623413, |
| "learning_rate": 3.9733579891788235e-05, |
| "loss": 0.0252, |
| "step": 6000 |
| }, |
| { |
| "epoch": 3.3384694401643555, |
| "grad_norm": 0.06139238923788071, |
| "learning_rate": 3.8877474145606465e-05, |
| "loss": 0.0244, |
| "step": 6500 |
| }, |
| { |
| "epoch": 3.5952747817154598, |
| "grad_norm": 0.09731902182102203, |
| "learning_rate": 3.80213683994247e-05, |
| "loss": 0.0244, |
| "step": 7000 |
| }, |
| { |
| "epoch": 3.852080123266564, |
| "grad_norm": 0.07361181080341339, |
| "learning_rate": 3.716526265324293e-05, |
| "loss": 0.0242, |
| "step": 7500 |
| }, |
| { |
| "epoch": 4.108885464817668, |
| "grad_norm": 0.05958514288067818, |
| "learning_rate": 3.630915690706116e-05, |
| "loss": 0.0233, |
| "step": 8000 |
| }, |
| { |
| "epoch": 4.3656908063687725, |
| "grad_norm": 0.1113390326499939, |
| "learning_rate": 3.5453051160879394e-05, |
| "loss": 0.0224, |
| "step": 8500 |
| }, |
| { |
| "epoch": 4.622496147919877, |
| "grad_norm": 0.09913190454244614, |
| "learning_rate": 3.4596945414697624e-05, |
| "loss": 0.0227, |
| "step": 9000 |
| }, |
| { |
| "epoch": 4.879301489470981, |
| "grad_norm": 0.13124661147594452, |
| "learning_rate": 3.3740839668515855e-05, |
| "loss": 0.0221, |
| "step": 9500 |
| }, |
| { |
| "epoch": 5.136106831022086, |
| "grad_norm": 0.09466934204101562, |
| "learning_rate": 3.2884733922334085e-05, |
| "loss": 0.0215, |
| "step": 10000 |
| }, |
| { |
| "epoch": 5.136106831022086, |
| "eval_loss": 0.026011742651462555, |
| "eval_runtime": 181.145, |
| "eval_samples_per_second": 14.326, |
| "eval_steps_per_second": 3.583, |
| "step": 10000 |
| }, |
| { |
| "epoch": 5.3929121725731894, |
| "grad_norm": 0.13241207599639893, |
| "learning_rate": 3.202862817615232e-05, |
| "loss": 0.0203, |
| "step": 10500 |
| }, |
| { |
| "epoch": 5.649717514124294, |
| "grad_norm": 0.10956753045320511, |
| "learning_rate": 3.117252242997055e-05, |
| "loss": 0.0205, |
| "step": 11000 |
| }, |
| { |
| "epoch": 5.906522855675398, |
| "grad_norm": 0.1889301985502243, |
| "learning_rate": 3.0316416683788784e-05, |
| "loss": 0.0206, |
| "step": 11500 |
| }, |
| { |
| "epoch": 6.163328197226503, |
| "grad_norm": 0.12303808331489563, |
| "learning_rate": 2.9460310937607018e-05, |
| "loss": 0.019, |
| "step": 12000 |
| }, |
| { |
| "epoch": 6.420133538777606, |
| "grad_norm": 0.13128505647182465, |
| "learning_rate": 2.8604205191425248e-05, |
| "loss": 0.0189, |
| "step": 12500 |
| }, |
| { |
| "epoch": 6.676938880328711, |
| "grad_norm": 0.23012486100196838, |
| "learning_rate": 2.7748099445243475e-05, |
| "loss": 0.0176, |
| "step": 13000 |
| }, |
| { |
| "epoch": 6.933744221879815, |
| "grad_norm": 0.16152064502239227, |
| "learning_rate": 2.6891993699061706e-05, |
| "loss": 0.0186, |
| "step": 13500 |
| }, |
| { |
| "epoch": 7.1905495634309196, |
| "grad_norm": 0.11679325997829437, |
| "learning_rate": 2.603588795287994e-05, |
| "loss": 0.0176, |
| "step": 14000 |
| }, |
| { |
| "epoch": 7.447354904982023, |
| "grad_norm": 0.21249671280384064, |
| "learning_rate": 2.517978220669817e-05, |
| "loss": 0.0166, |
| "step": 14500 |
| }, |
| { |
| "epoch": 7.704160246533128, |
| "grad_norm": 0.18364188075065613, |
| "learning_rate": 2.4323676460516404e-05, |
| "loss": 0.0166, |
| "step": 15000 |
| }, |
| { |
| "epoch": 7.704160246533128, |
| "eval_loss": 0.03057297319173813, |
| "eval_runtime": 185.7853, |
| "eval_samples_per_second": 13.968, |
| "eval_steps_per_second": 3.493, |
| "step": 15000 |
| }, |
| { |
| "epoch": 7.960965588084232, |
| "grad_norm": 0.2111167013645172, |
| "learning_rate": 2.3467570714334635e-05, |
| "loss": 0.0167, |
| "step": 15500 |
| }, |
| { |
| "epoch": 8.217770929635336, |
| "grad_norm": 0.3737495243549347, |
| "learning_rate": 2.261146496815287e-05, |
| "loss": 0.0147, |
| "step": 16000 |
| }, |
| { |
| "epoch": 8.474576271186441, |
| "grad_norm": 0.2299918681383133, |
| "learning_rate": 2.17553592219711e-05, |
| "loss": 0.0145, |
| "step": 16500 |
| }, |
| { |
| "epoch": 8.731381612737545, |
| "grad_norm": 0.151510551571846, |
| "learning_rate": 2.089925347578933e-05, |
| "loss": 0.0143, |
| "step": 17000 |
| }, |
| { |
| "epoch": 8.988186954288649, |
| "grad_norm": 0.16585668921470642, |
| "learning_rate": 2.004314772960756e-05, |
| "loss": 0.015, |
| "step": 17500 |
| }, |
| { |
| "epoch": 9.244992295839754, |
| "grad_norm": 0.14493027329444885, |
| "learning_rate": 1.9187041983425794e-05, |
| "loss": 0.0133, |
| "step": 18000 |
| }, |
| { |
| "epoch": 9.501797637390858, |
| "grad_norm": 0.30722230672836304, |
| "learning_rate": 1.8330936237244024e-05, |
| "loss": 0.0129, |
| "step": 18500 |
| }, |
| { |
| "epoch": 9.758602978941962, |
| "grad_norm": 0.09673864394426346, |
| "learning_rate": 1.747483049106226e-05, |
| "loss": 0.0127, |
| "step": 19000 |
| }, |
| { |
| "epoch": 10.015408320493066, |
| "grad_norm": 0.2731100916862488, |
| "learning_rate": 1.661872474488049e-05, |
| "loss": 0.013, |
| "step": 19500 |
| }, |
| { |
| "epoch": 10.272213662044171, |
| "grad_norm": 0.11305446922779083, |
| "learning_rate": 1.576261899869872e-05, |
| "loss": 0.0112, |
| "step": 20000 |
| }, |
| { |
| "epoch": 10.272213662044171, |
| "eval_loss": 0.042304039001464844, |
| "eval_runtime": 182.015, |
| "eval_samples_per_second": 14.257, |
| "eval_steps_per_second": 3.566, |
| "step": 20000 |
| }, |
| { |
| "epoch": 10.529019003595275, |
| "grad_norm": 0.30115944147109985, |
| "learning_rate": 1.4906513252516952e-05, |
| "loss": 0.0118, |
| "step": 20500 |
| }, |
| { |
| "epoch": 10.785824345146379, |
| "grad_norm": 0.22803886234760284, |
| "learning_rate": 1.4050407506335184e-05, |
| "loss": 0.011, |
| "step": 21000 |
| }, |
| { |
| "epoch": 11.042629686697483, |
| "grad_norm": 0.08788817375898361, |
| "learning_rate": 1.3194301760153416e-05, |
| "loss": 0.0114, |
| "step": 21500 |
| }, |
| { |
| "epoch": 11.299435028248588, |
| "grad_norm": 0.47961196303367615, |
| "learning_rate": 1.2338196013971646e-05, |
| "loss": 0.0102, |
| "step": 22000 |
| }, |
| { |
| "epoch": 11.556240369799692, |
| "grad_norm": 0.3627168834209442, |
| "learning_rate": 1.1482090267789877e-05, |
| "loss": 0.01, |
| "step": 22500 |
| }, |
| { |
| "epoch": 11.813045711350796, |
| "grad_norm": 0.22580872476100922, |
| "learning_rate": 1.062598452160811e-05, |
| "loss": 0.0103, |
| "step": 23000 |
| }, |
| { |
| "epoch": 12.0698510529019, |
| "grad_norm": 0.26205044984817505, |
| "learning_rate": 9.769878775426341e-06, |
| "loss": 0.0098, |
| "step": 23500 |
| }, |
| { |
| "epoch": 12.326656394453005, |
| "grad_norm": 0.16040152311325073, |
| "learning_rate": 8.913773029244572e-06, |
| "loss": 0.0093, |
| "step": 24000 |
| }, |
| { |
| "epoch": 12.583461736004109, |
| "grad_norm": 0.1721157431602478, |
| "learning_rate": 8.057667283062804e-06, |
| "loss": 0.0093, |
| "step": 24500 |
| }, |
| { |
| "epoch": 12.840267077555213, |
| "grad_norm": 0.583092212677002, |
| "learning_rate": 7.201561536881036e-06, |
| "loss": 0.0088, |
| "step": 25000 |
| }, |
| { |
| "epoch": 12.840267077555213, |
| "eval_loss": 0.05068250373005867, |
| "eval_runtime": 180.1951, |
| "eval_samples_per_second": 14.401, |
| "eval_steps_per_second": 3.602, |
| "step": 25000 |
| } |
| ], |
| "logging_steps": 500, |
| "max_steps": 29205, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 15, |
| "save_steps": 5000, |
| "stateful_callbacks": { |
| "EarlyStoppingCallback": { |
| "args": { |
| "early_stopping_patience": 3, |
| "early_stopping_threshold": 0.0 |
| }, |
| "attributes": { |
| "early_stopping_patience_counter": 3 |
| } |
| }, |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 2.16426787897344e+17, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|