| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 2.0, |
| "eval_steps": 500, |
| "global_step": 278, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.03597122302158273, |
| "grad_norm": 0.09922864826944784, |
| "learning_rate": 4.4444444444444447e-05, |
| "loss": 0.9273451805114746, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.07194244604316546, |
| "grad_norm": 0.09775713327362429, |
| "learning_rate": 0.0001, |
| "loss": 0.879729175567627, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.1079136690647482, |
| "grad_norm": 0.06962031038398203, |
| "learning_rate": 9.991477798614638e-05, |
| "loss": 0.8456396102905274, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.14388489208633093, |
| "grad_norm": 0.07512256866077939, |
| "learning_rate": 9.965940245625131e-05, |
| "loss": 0.8059403419494628, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.17985611510791366, |
| "grad_norm": 0.06345996636264109, |
| "learning_rate": 9.923474395499265e-05, |
| "loss": 0.7759585857391358, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.2158273381294964, |
| "grad_norm": 0.06004270955330792, |
| "learning_rate": 9.864225009247751e-05, |
| "loss": 0.7483164310455322, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.2517985611510791, |
| "grad_norm": 0.07437488371254487, |
| "learning_rate": 9.788394060951229e-05, |
| "loss": 0.8478424072265625, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.28776978417266186, |
| "grad_norm": 0.07975161331683868, |
| "learning_rate": 9.696240049254743e-05, |
| "loss": 0.8275875091552735, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.3237410071942446, |
| "grad_norm": 0.056110204238877824, |
| "learning_rate": 9.588077116176756e-05, |
| "loss": 0.745372200012207, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.3597122302158273, |
| "grad_norm": 0.07311021819090241, |
| "learning_rate": 9.464273976236517e-05, |
| "loss": 0.8448892593383789, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.39568345323741005, |
| "grad_norm": 0.06206181655606858, |
| "learning_rate": 9.325252659550309e-05, |
| "loss": 0.803032398223877, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.4316546762589928, |
| "grad_norm": 0.07330787459721261, |
| "learning_rate": 9.171487073181198e-05, |
| "loss": 0.7427041053771972, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.4676258992805755, |
| "grad_norm": 0.06690776136899469, |
| "learning_rate": 9.003501385646449e-05, |
| "loss": 0.845458984375, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.5035971223021583, |
| "grad_norm": 0.06724649328703676, |
| "learning_rate": 8.821868240089676e-05, |
| "loss": 0.7568838119506835, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.539568345323741, |
| "grad_norm": 0.07423296624528276, |
| "learning_rate": 8.62720680220876e-05, |
| "loss": 0.8067834854125977, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.5755395683453237, |
| "grad_norm": 0.06821297058376587, |
| "learning_rate": 8.420180649593929e-05, |
| "loss": 0.8216766357421875, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.6115107913669064, |
| "grad_norm": 0.0845331853669141, |
| "learning_rate": 8.201495509671037e-05, |
| "loss": 0.7999070167541504, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.6474820143884892, |
| "grad_norm": 0.08940520339956634, |
| "learning_rate": 7.971896853961042e-05, |
| "loss": 0.7463856220245362, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.6834532374100719, |
| "grad_norm": 0.0739578639152744, |
| "learning_rate": 7.732167356856655e-05, |
| "loss": 0.7730668067932129, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.7194244604316546, |
| "grad_norm": 0.0695624845552875, |
| "learning_rate": 7.483124227578811e-05, |
| "loss": 0.732236385345459, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.7553956834532374, |
| "grad_norm": 0.08718202167856036, |
| "learning_rate": 7.225616424408045e-05, |
| "loss": 0.8057218551635742, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.7913669064748201, |
| "grad_norm": 0.06252924097123615, |
| "learning_rate": 6.960521760687129e-05, |
| "loss": 0.7559296131134033, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.8273381294964028, |
| "grad_norm": 0.0683171512398649, |
| "learning_rate": 6.688743912460229e-05, |
| "loss": 0.798275089263916, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.8633093525179856, |
| "grad_norm": 0.06946370954099755, |
| "learning_rate": 6.411209337949214e-05, |
| "loss": 0.7671150684356689, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.8992805755395683, |
| "grad_norm": 0.06965417114125676, |
| "learning_rate": 6.128864119368234e-05, |
| "loss": 0.6539465904235839, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.935251798561151, |
| "grad_norm": 0.06667373525748503, |
| "learning_rate": 5.8426707378424675e-05, |
| "loss": 0.8279660224914551, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.9712230215827338, |
| "grad_norm": 0.06288374976421086, |
| "learning_rate": 5.553604792424922e-05, |
| "loss": 0.8029556274414062, |
| "step": 135 |
| }, |
| { |
| "epoch": 1.0071942446043165, |
| "grad_norm": 0.07572492394881657, |
| "learning_rate": 5.262651674395799e-05, |
| "loss": 0.7947874069213867, |
| "step": 140 |
| }, |
| { |
| "epoch": 1.0431654676258992, |
| "grad_norm": 0.07014160285000554, |
| "learning_rate": 4.9708032081813144e-05, |
| "loss": 0.7853281021118164, |
| "step": 145 |
| }, |
| { |
| "epoch": 1.079136690647482, |
| "grad_norm": 0.07147065892903694, |
| "learning_rate": 4.679054270342703e-05, |
| "loss": 0.7561178207397461, |
| "step": 150 |
| }, |
| { |
| "epoch": 1.1151079136690647, |
| "grad_norm": 0.08320440539524798, |
| "learning_rate": 4.3883993981608576e-05, |
| "loss": 0.6941920757293701, |
| "step": 155 |
| }, |
| { |
| "epoch": 1.1510791366906474, |
| "grad_norm": 0.09370719660499001, |
| "learning_rate": 4.0998293993775237e-05, |
| "loss": 0.7089109420776367, |
| "step": 160 |
| }, |
| { |
| "epoch": 1.1870503597122302, |
| "grad_norm": 0.09052205189656633, |
| "learning_rate": 3.814327974650067e-05, |
| "loss": 0.7286513328552247, |
| "step": 165 |
| }, |
| { |
| "epoch": 1.223021582733813, |
| "grad_norm": 0.08062290558666846, |
| "learning_rate": 3.532868364233416e-05, |
| "loss": 0.7593664169311524, |
| "step": 170 |
| }, |
| { |
| "epoch": 1.2589928057553956, |
| "grad_norm": 0.09623790985860192, |
| "learning_rate": 3.2564100303203035e-05, |
| "loss": 0.8009302139282226, |
| "step": 175 |
| }, |
| { |
| "epoch": 1.2949640287769784, |
| "grad_norm": 0.13362882285179445, |
| "learning_rate": 2.9858953863492334e-05, |
| "loss": 0.7465702533721924, |
| "step": 180 |
| }, |
| { |
| "epoch": 1.330935251798561, |
| "grad_norm": 0.08708298515845442, |
| "learning_rate": 2.722246584429652e-05, |
| "loss": 0.7011271476745605, |
| "step": 185 |
| }, |
| { |
| "epoch": 1.3669064748201438, |
| "grad_norm": 0.09765847620975966, |
| "learning_rate": 2.4663623718355444e-05, |
| "loss": 0.6933589935302734, |
| "step": 190 |
| }, |
| { |
| "epoch": 1.4028776978417266, |
| "grad_norm": 0.11061497330176863, |
| "learning_rate": 2.219115027283339e-05, |
| "loss": 0.6514883041381836, |
| "step": 195 |
| }, |
| { |
| "epoch": 1.4388489208633093, |
| "grad_norm": 0.13003536918832742, |
| "learning_rate": 1.9813473874379395e-05, |
| "loss": 0.7087651252746582, |
| "step": 200 |
| }, |
| { |
| "epoch": 1.474820143884892, |
| "grad_norm": 0.12215362417494327, |
| "learning_rate": 1.7538699737832236e-05, |
| "loss": 0.6702914714813233, |
| "step": 205 |
| }, |
| { |
| "epoch": 1.5107913669064748, |
| "grad_norm": 0.10495077293191406, |
| "learning_rate": 1.5374582296511053e-05, |
| "loss": 0.7487486839294434, |
| "step": 210 |
| }, |
| { |
| "epoch": 1.5467625899280577, |
| "grad_norm": 0.13424475386206883, |
| "learning_rate": 1.332849876827842e-05, |
| "loss": 0.7264563560485839, |
| "step": 215 |
| }, |
| { |
| "epoch": 1.5827338129496402, |
| "grad_norm": 0.12241730701789537, |
| "learning_rate": 1.1407424007485929e-05, |
| "loss": 0.6742488861083984, |
| "step": 220 |
| }, |
| { |
| "epoch": 1.6187050359712232, |
| "grad_norm": 0.10964031951322462, |
| "learning_rate": 9.61790672852868e-06, |
| "loss": 0.7405488967895508, |
| "step": 225 |
| }, |
| { |
| "epoch": 1.6546762589928057, |
| "grad_norm": 0.11177149802046989, |
| "learning_rate": 7.966047182060226e-06, |
| "loss": 0.7087203979492187, |
| "step": 230 |
| }, |
| { |
| "epoch": 1.6906474820143886, |
| "grad_norm": 0.1218003930872436, |
| "learning_rate": 6.4574763599666856e-06, |
| "loss": 0.705730390548706, |
| "step": 235 |
| }, |
| { |
| "epoch": 1.7266187050359711, |
| "grad_norm": 0.10471844065341074, |
| "learning_rate": 5.097336799988067e-06, |
| "loss": 0.6330158710479736, |
| "step": 240 |
| }, |
| { |
| "epoch": 1.762589928057554, |
| "grad_norm": 0.1036431593783518, |
| "learning_rate": 3.890265055421283e-06, |
| "loss": 0.6806740760803223, |
| "step": 245 |
| }, |
| { |
| "epoch": 1.7985611510791366, |
| "grad_norm": 0.10616735376524208, |
| "learning_rate": 2.840375889663871e-06, |
| "loss": 0.7142208099365235, |
| "step": 250 |
| }, |
| { |
| "epoch": 1.8345323741007196, |
| "grad_norm": 0.11988724859801742, |
| "learning_rate": 1.9512482494769613e-06, |
| "loss": 0.7671703338623047, |
| "step": 255 |
| }, |
| { |
| "epoch": 1.870503597122302, |
| "grad_norm": 0.11871032303539183, |
| "learning_rate": 1.2259130647833627e-06, |
| "loss": 0.7068417072296143, |
| "step": 260 |
| }, |
| { |
| "epoch": 1.906474820143885, |
| "grad_norm": 0.11598204652847133, |
| "learning_rate": 6.668429165893997e-07, |
| "loss": 0.6564496994018555, |
| "step": 265 |
| }, |
| { |
| "epoch": 1.9424460431654675, |
| "grad_norm": 0.11993913393449739, |
| "learning_rate": 2.7594360825166644e-07, |
| "loss": 0.727387523651123, |
| "step": 270 |
| }, |
| { |
| "epoch": 1.9784172661870505, |
| "grad_norm": 0.10081443145587521, |
| "learning_rate": 5.454766882097007e-08, |
| "loss": 0.7604126453399658, |
| "step": 275 |
| }, |
| { |
| "epoch": 2.0, |
| "step": 278, |
| "total_flos": 1037690204061696.0, |
| "train_loss": 0.7568303492429446, |
| "train_runtime": 5056.394, |
| "train_samples_per_second": 0.439, |
| "train_steps_per_second": 0.055 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 278, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 2, |
| "save_steps": 100, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1037690204061696.0, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|