| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 50, |
| "global_step": 71, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.014222222222222223, |
| "grad_norm": 0.10009177774190903, |
| "learning_rate": 0.0, |
| "loss": 0.704937219619751, |
| "step": 1 |
| }, |
| { |
| "epoch": 0.028444444444444446, |
| "grad_norm": 0.10451912879943848, |
| "learning_rate": 4.5454545454545457e-07, |
| "loss": 0.7318670749664307, |
| "step": 2 |
| }, |
| { |
| "epoch": 0.042666666666666665, |
| "grad_norm": 0.07928935438394547, |
| "learning_rate": 9.090909090909091e-07, |
| "loss": 0.6017104387283325, |
| "step": 3 |
| }, |
| { |
| "epoch": 0.05688888888888889, |
| "grad_norm": 0.048069290816783905, |
| "learning_rate": 1.3636363636363636e-06, |
| "loss": 0.6029032468795776, |
| "step": 4 |
| }, |
| { |
| "epoch": 0.07111111111111111, |
| "grad_norm": 0.02866927720606327, |
| "learning_rate": 1.8181818181818183e-06, |
| "loss": 0.5051919221878052, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.08533333333333333, |
| "grad_norm": 0.022201435640454292, |
| "learning_rate": 2.2727272727272728e-06, |
| "loss": 0.5942293405532837, |
| "step": 6 |
| }, |
| { |
| "epoch": 0.09955555555555555, |
| "grad_norm": 0.018263086676597595, |
| "learning_rate": 2.7272727272727272e-06, |
| "loss": 0.5781338214874268, |
| "step": 7 |
| }, |
| { |
| "epoch": 0.11377777777777778, |
| "grad_norm": 0.01827927492558956, |
| "learning_rate": 3.181818181818182e-06, |
| "loss": 0.5180214047431946, |
| "step": 8 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 0.014687157236039639, |
| "learning_rate": 3.6363636363636366e-06, |
| "loss": 0.5420116186141968, |
| "step": 9 |
| }, |
| { |
| "epoch": 0.14222222222222222, |
| "grad_norm": 0.012042663060128689, |
| "learning_rate": 4.0909090909090915e-06, |
| "loss": 0.5345908403396606, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.15644444444444444, |
| "grad_norm": 0.012934447266161442, |
| "learning_rate": 4.5454545454545455e-06, |
| "loss": 0.46689414978027344, |
| "step": 11 |
| }, |
| { |
| "epoch": 0.17066666666666666, |
| "grad_norm": 0.012295619584619999, |
| "learning_rate": 5e-06, |
| "loss": 0.5062131881713867, |
| "step": 12 |
| }, |
| { |
| "epoch": 0.18488888888888888, |
| "grad_norm": 0.011856353841722012, |
| "learning_rate": 5.4545454545454545e-06, |
| "loss": 0.5063096880912781, |
| "step": 13 |
| }, |
| { |
| "epoch": 0.1991111111111111, |
| "grad_norm": 0.009451798163354397, |
| "learning_rate": 5.90909090909091e-06, |
| "loss": 0.5130602121353149, |
| "step": 14 |
| }, |
| { |
| "epoch": 0.21333333333333335, |
| "grad_norm": 0.01111016795039177, |
| "learning_rate": 6.363636363636364e-06, |
| "loss": 0.5016045570373535, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.22755555555555557, |
| "grad_norm": 0.010735263116657734, |
| "learning_rate": 6.818181818181818e-06, |
| "loss": 0.49827253818511963, |
| "step": 16 |
| }, |
| { |
| "epoch": 0.24177777777777779, |
| "grad_norm": 0.010710126720368862, |
| "learning_rate": 7.272727272727273e-06, |
| "loss": 0.479993999004364, |
| "step": 17 |
| }, |
| { |
| "epoch": 0.256, |
| "grad_norm": 0.009600787423551083, |
| "learning_rate": 7.727272727272727e-06, |
| "loss": 0.4925128221511841, |
| "step": 18 |
| }, |
| { |
| "epoch": 0.2702222222222222, |
| "grad_norm": 0.010486789979040623, |
| "learning_rate": 8.181818181818183e-06, |
| "loss": 0.5188871622085571, |
| "step": 19 |
| }, |
| { |
| "epoch": 0.28444444444444444, |
| "grad_norm": 0.013396799564361572, |
| "learning_rate": 8.636363636363637e-06, |
| "loss": 0.5192517042160034, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.2986666666666667, |
| "grad_norm": 0.009299715980887413, |
| "learning_rate": 9.090909090909091e-06, |
| "loss": 0.4580157399177551, |
| "step": 21 |
| }, |
| { |
| "epoch": 0.3128888888888889, |
| "grad_norm": 0.009691477753221989, |
| "learning_rate": 9.545454545454547e-06, |
| "loss": 0.49455106258392334, |
| "step": 22 |
| }, |
| { |
| "epoch": 0.32711111111111113, |
| "grad_norm": 0.008896945044398308, |
| "learning_rate": 1e-05, |
| "loss": 0.48672014474868774, |
| "step": 23 |
| }, |
| { |
| "epoch": 0.3413333333333333, |
| "grad_norm": 0.009234798140823841, |
| "learning_rate": 9.999323662872998e-06, |
| "loss": 0.5052967071533203, |
| "step": 24 |
| }, |
| { |
| "epoch": 0.35555555555555557, |
| "grad_norm": 0.009167873300611973, |
| "learning_rate": 9.99729483446475e-06, |
| "loss": 0.4740435779094696, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.36977777777777776, |
| "grad_norm": 0.009001555852591991, |
| "learning_rate": 9.993914063644053e-06, |
| "loss": 0.47315946221351624, |
| "step": 26 |
| }, |
| { |
| "epoch": 0.384, |
| "grad_norm": 0.00915393978357315, |
| "learning_rate": 9.989182265027232e-06, |
| "loss": 0.4557237923145294, |
| "step": 27 |
| }, |
| { |
| "epoch": 0.3982222222222222, |
| "grad_norm": 0.008505849167704582, |
| "learning_rate": 9.98310071873072e-06, |
| "loss": 0.46317678689956665, |
| "step": 28 |
| }, |
| { |
| "epoch": 0.41244444444444445, |
| "grad_norm": 0.008668503724038601, |
| "learning_rate": 9.975671070024741e-06, |
| "loss": 0.4533114433288574, |
| "step": 29 |
| }, |
| { |
| "epoch": 0.4266666666666667, |
| "grad_norm": 0.009284070692956448, |
| "learning_rate": 9.966895328888195e-06, |
| "loss": 0.5624005794525146, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.4408888888888889, |
| "grad_norm": 0.00935022346675396, |
| "learning_rate": 9.956775869464901e-06, |
| "loss": 0.4420597553253174, |
| "step": 31 |
| }, |
| { |
| "epoch": 0.45511111111111113, |
| "grad_norm": 0.008682326413691044, |
| "learning_rate": 9.945315429421307e-06, |
| "loss": 0.4745720624923706, |
| "step": 32 |
| }, |
| { |
| "epoch": 0.4693333333333333, |
| "grad_norm": 0.007816251367330551, |
| "learning_rate": 9.932517109205849e-06, |
| "loss": 0.439736545085907, |
| "step": 33 |
| }, |
| { |
| "epoch": 0.48355555555555557, |
| "grad_norm": 0.009315130300819874, |
| "learning_rate": 9.918384371210178e-06, |
| "loss": 0.5357447862625122, |
| "step": 34 |
| }, |
| { |
| "epoch": 0.49777777777777776, |
| "grad_norm": 0.009141312912106514, |
| "learning_rate": 9.902921038832456e-06, |
| "loss": 0.519463062286377, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.512, |
| "grad_norm": 0.008826862089335918, |
| "learning_rate": 9.886131295443003e-06, |
| "loss": 0.48377716541290283, |
| "step": 36 |
| }, |
| { |
| "epoch": 0.5262222222222223, |
| "grad_norm": 0.008700878359377384, |
| "learning_rate": 9.868019683252543e-06, |
| "loss": 0.48436683416366577, |
| "step": 37 |
| }, |
| { |
| "epoch": 0.5404444444444444, |
| "grad_norm": 0.011412105523049831, |
| "learning_rate": 9.848591102083375e-06, |
| "loss": 0.4949312210083008, |
| "step": 38 |
| }, |
| { |
| "epoch": 0.5546666666666666, |
| "grad_norm": 0.009214675053954124, |
| "learning_rate": 9.82785080804381e-06, |
| "loss": 0.4901079535484314, |
| "step": 39 |
| }, |
| { |
| "epoch": 0.5688888888888889, |
| "grad_norm": 0.01163570862263441, |
| "learning_rate": 9.805804412106197e-06, |
| "loss": 0.6070311665534973, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.5831111111111111, |
| "grad_norm": 0.009325762279331684, |
| "learning_rate": 9.782457878588977e-06, |
| "loss": 0.48412975668907166, |
| "step": 41 |
| }, |
| { |
| "epoch": 0.5973333333333334, |
| "grad_norm": 0.008266489021480083, |
| "learning_rate": 9.75781752354311e-06, |
| "loss": 0.4440661072731018, |
| "step": 42 |
| }, |
| { |
| "epoch": 0.6115555555555555, |
| "grad_norm": 0.00843313243240118, |
| "learning_rate": 9.731890013043367e-06, |
| "loss": 0.502442479133606, |
| "step": 43 |
| }, |
| { |
| "epoch": 0.6257777777777778, |
| "grad_norm": 0.00934070535004139, |
| "learning_rate": 9.704682361384941e-06, |
| "loss": 0.47924768924713135, |
| "step": 44 |
| }, |
| { |
| "epoch": 0.64, |
| "grad_norm": 0.010360248386859894, |
| "learning_rate": 9.676201929185809e-06, |
| "loss": 0.4161364436149597, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.6542222222222223, |
| "grad_norm": 0.0094041982665658, |
| "learning_rate": 9.646456421395447e-06, |
| "loss": 0.48306071758270264, |
| "step": 46 |
| }, |
| { |
| "epoch": 0.6684444444444444, |
| "grad_norm": 0.009070150554180145, |
| "learning_rate": 9.615453885210368e-06, |
| "loss": 0.4774249792098999, |
| "step": 47 |
| }, |
| { |
| "epoch": 0.6826666666666666, |
| "grad_norm": 0.009771023876965046, |
| "learning_rate": 9.583202707897075e-06, |
| "loss": 0.4948921203613281, |
| "step": 48 |
| }, |
| { |
| "epoch": 0.6968888888888889, |
| "grad_norm": 0.008749130181968212, |
| "learning_rate": 9.549711614523007e-06, |
| "loss": 0.4825429916381836, |
| "step": 49 |
| }, |
| { |
| "epoch": 0.7111111111111111, |
| "grad_norm": 0.007630312815308571, |
| "learning_rate": 9.514989665596114e-06, |
| "loss": 0.46055319905281067, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.7111111111111111, |
| "eval_loss": 0.48781517148017883, |
| "eval_runtime": 4.5903, |
| "eval_samples_per_second": 43.57, |
| "eval_steps_per_second": 21.785, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.7253333333333334, |
| "grad_norm": 0.008408056572079659, |
| "learning_rate": 9.479046254613673e-06, |
| "loss": 0.4626312255859375, |
| "step": 51 |
| }, |
| { |
| "epoch": 0.7395555555555555, |
| "grad_norm": 0.007699670735746622, |
| "learning_rate": 9.441891105521005e-06, |
| "loss": 0.4499070644378662, |
| "step": 52 |
| }, |
| { |
| "epoch": 0.7537777777777778, |
| "grad_norm": 0.009070169180631638, |
| "learning_rate": 9.40353427008083e-06, |
| "loss": 0.4825657606124878, |
| "step": 53 |
| }, |
| { |
| "epoch": 0.768, |
| "grad_norm": 0.007915646769106388, |
| "learning_rate": 9.3639861251539e-06, |
| "loss": 0.4534468650817871, |
| "step": 54 |
| }, |
| { |
| "epoch": 0.7822222222222223, |
| "grad_norm": 0.011336776427924633, |
| "learning_rate": 9.323257369891702e-06, |
| "loss": 0.4988631010055542, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.7964444444444444, |
| "grad_norm": 0.00761911366134882, |
| "learning_rate": 9.281359022841966e-06, |
| "loss": 0.4371437728404999, |
| "step": 56 |
| }, |
| { |
| "epoch": 0.8106666666666666, |
| "grad_norm": 0.009316994808614254, |
| "learning_rate": 9.238302418967757e-06, |
| "loss": 0.4956932067871094, |
| "step": 57 |
| }, |
| { |
| "epoch": 0.8248888888888889, |
| "grad_norm": 0.009821794927120209, |
| "learning_rate": 9.194099206580981e-06, |
| "loss": 0.5392974615097046, |
| "step": 58 |
| }, |
| { |
| "epoch": 0.8391111111111111, |
| "grad_norm": 0.008422416634857655, |
| "learning_rate": 9.14876134419111e-06, |
| "loss": 0.47983357310295105, |
| "step": 59 |
| }, |
| { |
| "epoch": 0.8533333333333334, |
| "grad_norm": 0.00837004091590643, |
| "learning_rate": 9.102301097269974e-06, |
| "loss": 0.5183510780334473, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.8675555555555555, |
| "grad_norm": 0.008596297353506088, |
| "learning_rate": 9.05473103493355e-06, |
| "loss": 0.47292786836624146, |
| "step": 61 |
| }, |
| { |
| "epoch": 0.8817777777777778, |
| "grad_norm": 0.009059637784957886, |
| "learning_rate": 9.006064026541549e-06, |
| "loss": 0.5188573598861694, |
| "step": 62 |
| }, |
| { |
| "epoch": 0.896, |
| "grad_norm": 0.007816917262971401, |
| "learning_rate": 8.956313238215824e-06, |
| "loss": 0.45641395449638367, |
| "step": 63 |
| }, |
| { |
| "epoch": 0.9102222222222223, |
| "grad_norm": 0.007985202595591545, |
| "learning_rate": 8.905492129278478e-06, |
| "loss": 0.4586361050605774, |
| "step": 64 |
| }, |
| { |
| "epoch": 0.9244444444444444, |
| "grad_norm": 0.007170964498072863, |
| "learning_rate": 8.85361444861063e-06, |
| "loss": 0.4394378066062927, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.9386666666666666, |
| "grad_norm": 0.00909162126481533, |
| "learning_rate": 8.800694230932885e-06, |
| "loss": 0.509820818901062, |
| "step": 66 |
| }, |
| { |
| "epoch": 0.9528888888888889, |
| "grad_norm": 0.00786044355481863, |
| "learning_rate": 8.74674579300843e-06, |
| "loss": 0.43584880232810974, |
| "step": 67 |
| }, |
| { |
| "epoch": 0.9671111111111111, |
| "grad_norm": 0.00832363311201334, |
| "learning_rate": 8.691783729769874e-06, |
| "loss": 0.4317135214805603, |
| "step": 68 |
| }, |
| { |
| "epoch": 0.9813333333333333, |
| "grad_norm": 0.008376957848668098, |
| "learning_rate": 8.635822910370793e-06, |
| "loss": 0.49933016300201416, |
| "step": 69 |
| }, |
| { |
| "epoch": 0.9955555555555555, |
| "grad_norm": 0.00835558120161295, |
| "learning_rate": 8.578878474163115e-06, |
| "loss": 0.46570393443107605, |
| "step": 70 |
| }, |
| { |
| "epoch": 1.0, |
| "grad_norm": 0.013256405480206013, |
| "learning_rate": 8.520965826601394e-06, |
| "loss": 0.45107918977737427, |
| "step": 71 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 213, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1.6226691949002752e+16, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|