{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 5.0, "eval_steps": 500, "global_step": 90, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.05555555555555555, "grad_norm": 17.469547271728516, "learning_rate": 0.0, "loss": 0.732, "step": 1 }, { "epoch": 0.1111111111111111, "grad_norm": 22.75277328491211, "learning_rate": 2.2222222222222223e-05, "loss": 0.8724, "step": 2 }, { "epoch": 0.16666666666666666, "grad_norm": 23.468505859375, "learning_rate": 4.4444444444444447e-05, "loss": 1.109, "step": 3 }, { "epoch": 0.2222222222222222, "grad_norm": 7.13322639465332, "learning_rate": 6.666666666666667e-05, "loss": 0.4281, "step": 4 }, { "epoch": 0.2777777777777778, "grad_norm": 8.956477165222168, "learning_rate": 8.888888888888889e-05, "loss": 0.6818, "step": 5 }, { "epoch": 0.3333333333333333, "grad_norm": 4.624927520751953, "learning_rate": 0.00011111111111111112, "loss": 0.4782, "step": 6 }, { "epoch": 0.3888888888888889, "grad_norm": 10.363118171691895, "learning_rate": 0.00013333333333333334, "loss": 0.583, "step": 7 }, { "epoch": 0.4444444444444444, "grad_norm": 13.19287109375, "learning_rate": 0.00015555555555555556, "loss": 0.1829, "step": 8 }, { "epoch": 0.5, "grad_norm": 7.12847900390625, "learning_rate": 0.00017777777777777779, "loss": 0.2936, "step": 9 }, { "epoch": 0.5555555555555556, "grad_norm": 6.902624607086182, "learning_rate": 0.0002, "loss": 0.3713, "step": 10 }, { "epoch": 0.6111111111111112, "grad_norm": 3.943729877471924, "learning_rate": 0.00019992479525042303, "loss": 0.1013, "step": 11 }, { "epoch": 0.6666666666666666, "grad_norm": 4.228763580322266, "learning_rate": 0.0001996992941167792, "loss": 0.2174, "step": 12 }, { "epoch": 0.7222222222222222, "grad_norm": 3.2873263359069824, "learning_rate": 0.00019932383577419432, "loss": 0.2151, "step": 13 }, { "epoch": 0.7777777777777778, "grad_norm": 4.9496283531188965, "learning_rate": 0.00019879898494768093, "loss": 0.4807, "step": 14 }, { "epoch": 0.8333333333333334, "grad_norm": 5.166029930114746, "learning_rate": 0.00019812553106273847, "loss": 0.1647, "step": 15 }, { "epoch": 0.8888888888888888, "grad_norm": 8.569997787475586, "learning_rate": 0.00019730448705798239, "loss": 0.6876, "step": 16 }, { "epoch": 0.9444444444444444, "grad_norm": 2.3550405502319336, "learning_rate": 0.00019633708786158806, "loss": 0.1372, "step": 17 }, { "epoch": 1.0, "grad_norm": 3.146822214126587, "learning_rate": 0.00019522478853384155, "loss": 0.0177, "step": 18 }, { "epoch": 1.0555555555555556, "grad_norm": 1.8545197248458862, "learning_rate": 0.00019396926207859084, "loss": 0.0716, "step": 19 }, { "epoch": 1.1111111111111112, "grad_norm": 3.4503188133239746, "learning_rate": 0.00019257239692688907, "loss": 0.0969, "step": 20 }, { "epoch": 1.1666666666666667, "grad_norm": 0.7412665486335754, "learning_rate": 0.0001910362940966147, "loss": 0.0223, "step": 21 }, { "epoch": 1.2222222222222223, "grad_norm": 10.92074203491211, "learning_rate": 0.00018936326403234125, "loss": 0.2261, "step": 22 }, { "epoch": 1.2777777777777777, "grad_norm": 5.52816104888916, "learning_rate": 0.0001875558231302091, "loss": 0.2436, "step": 23 }, { "epoch": 1.3333333333333333, "grad_norm": 2.198107957839966, "learning_rate": 0.00018561668995302667, "loss": 0.061, "step": 24 }, { "epoch": 1.3888888888888888, "grad_norm": 3.341217041015625, "learning_rate": 0.00018354878114129367, "loss": 0.146, "step": 25 }, { "epoch": 1.4444444444444444, "grad_norm": 0.8276804089546204, "learning_rate": 0.00018135520702629675, "loss": 0.0242, "step": 26 }, { "epoch": 1.5, "grad_norm": 0.304651141166687, "learning_rate": 0.00017903926695187595, "loss": 0.0028, "step": 27 }, { "epoch": 1.5555555555555556, "grad_norm": 0.5654858946800232, "learning_rate": 0.0001766044443118978, "loss": 0.0058, "step": 28 }, { "epoch": 1.6111111111111112, "grad_norm": 0.1325332224369049, "learning_rate": 0.00017405440131090048, "loss": 0.0017, "step": 29 }, { "epoch": 1.6666666666666665, "grad_norm": 0.26076093316078186, "learning_rate": 0.00017139297345578994, "loss": 0.0028, "step": 30 }, { "epoch": 1.7222222222222223, "grad_norm": 7.41240930557251, "learning_rate": 0.0001686241637868734, "loss": 0.477, "step": 31 }, { "epoch": 1.7777777777777777, "grad_norm": 4.4489665031433105, "learning_rate": 0.0001657521368569064, "loss": 0.0604, "step": 32 }, { "epoch": 1.8333333333333335, "grad_norm": 4.5502166748046875, "learning_rate": 0.00016278121246720987, "loss": 0.2863, "step": 33 }, { "epoch": 1.8888888888888888, "grad_norm": 7.6348137855529785, "learning_rate": 0.00015971585917027862, "loss": 0.1511, "step": 34 }, { "epoch": 1.9444444444444444, "grad_norm": 11.212128639221191, "learning_rate": 0.00015656068754865387, "loss": 0.1643, "step": 35 }, { "epoch": 2.0, "grad_norm": 1.7633869647979736, "learning_rate": 0.00015332044328016914, "loss": 0.0374, "step": 36 }, { "epoch": 2.0555555555555554, "grad_norm": 0.567506730556488, "learning_rate": 0.00015000000000000001, "loss": 0.0055, "step": 37 }, { "epoch": 2.111111111111111, "grad_norm": 0.24216872453689575, "learning_rate": 0.0001466043519702539, "loss": 0.0045, "step": 38 }, { "epoch": 2.1666666666666665, "grad_norm": 6.3600053787231445, "learning_rate": 0.00014313860656812536, "loss": 0.1886, "step": 39 }, { "epoch": 2.2222222222222223, "grad_norm": 4.492520332336426, "learning_rate": 0.0001396079766039157, "loss": 0.053, "step": 40 }, { "epoch": 2.2777777777777777, "grad_norm": 1.1104881763458252, "learning_rate": 0.00013601777248047105, "loss": 0.0159, "step": 41 }, { "epoch": 2.3333333333333335, "grad_norm": 0.21907638013362885, "learning_rate": 0.00013237339420583212, "loss": 0.0018, "step": 42 }, { "epoch": 2.388888888888889, "grad_norm": 7.373753547668457, "learning_rate": 0.00012868032327110904, "loss": 0.1776, "step": 43 }, { "epoch": 2.4444444444444446, "grad_norm": 4.361786842346191, "learning_rate": 0.00012494411440579814, "loss": 0.1345, "step": 44 }, { "epoch": 2.5, "grad_norm": 3.8851583003997803, "learning_rate": 0.0001211703872229411, "loss": 0.1062, "step": 45 }, { "epoch": 2.5555555555555554, "grad_norm": 0.022876089438796043, "learning_rate": 0.00011736481776669306, "loss": 0.0004, "step": 46 }, { "epoch": 2.611111111111111, "grad_norm": 0.05538008362054825, "learning_rate": 0.00011353312997501313, "loss": 0.0008, "step": 47 }, { "epoch": 2.6666666666666665, "grad_norm": 0.5484911203384399, "learning_rate": 0.00010968108707031792, "loss": 0.0032, "step": 48 }, { "epoch": 2.7222222222222223, "grad_norm": 0.17594866454601288, "learning_rate": 0.00010581448289104758, "loss": 0.0029, "step": 49 }, { "epoch": 2.7777777777777777, "grad_norm": 0.42551591992378235, "learning_rate": 0.00010193913317718244, "loss": 0.0079, "step": 50 }, { "epoch": 2.8333333333333335, "grad_norm": 4.214639186859131, "learning_rate": 9.806086682281758e-05, "loss": 0.1177, "step": 51 }, { "epoch": 2.888888888888889, "grad_norm": 5.015491962432861, "learning_rate": 9.418551710895243e-05, "loss": 0.076, "step": 52 }, { "epoch": 2.9444444444444446, "grad_norm": 3.676525831222534, "learning_rate": 9.03189129296821e-05, "loss": 0.1516, "step": 53 }, { "epoch": 3.0, "grad_norm": 0.010427688248455524, "learning_rate": 8.646687002498692e-05, "loss": 0.0002, "step": 54 }, { "epoch": 3.0555555555555554, "grad_norm": 3.544339895248413, "learning_rate": 8.263518223330697e-05, "loss": 0.0387, "step": 55 }, { "epoch": 3.111111111111111, "grad_norm": 0.9848164319992065, "learning_rate": 7.882961277705895e-05, "loss": 0.0077, "step": 56 }, { "epoch": 3.1666666666666665, "grad_norm": 0.11404402554035187, "learning_rate": 7.505588559420189e-05, "loss": 0.0019, "step": 57 }, { "epoch": 3.2222222222222223, "grad_norm": 0.3479735255241394, "learning_rate": 7.131967672889101e-05, "loss": 0.0074, "step": 58 }, { "epoch": 3.2777777777777777, "grad_norm": 0.2785912752151489, "learning_rate": 6.762660579416791e-05, "loss": 0.0047, "step": 59 }, { "epoch": 3.3333333333333335, "grad_norm": 1.5189486742019653, "learning_rate": 6.398222751952899e-05, "loss": 0.0157, "step": 60 }, { "epoch": 3.388888888888889, "grad_norm": 1.2978746891021729, "learning_rate": 6.039202339608432e-05, "loss": 0.0148, "step": 61 }, { "epoch": 3.4444444444444446, "grad_norm": 0.19974800944328308, "learning_rate": 5.6861393431874675e-05, "loss": 0.0024, "step": 62 }, { "epoch": 3.5, "grad_norm": 0.08159318566322327, "learning_rate": 5.339564802974615e-05, "loss": 0.0011, "step": 63 }, { "epoch": 3.5555555555555554, "grad_norm": 0.034631360322237015, "learning_rate": 5.000000000000002e-05, "loss": 0.0005, "step": 64 }, { "epoch": 3.611111111111111, "grad_norm": 0.13056862354278564, "learning_rate": 4.66795567198309e-05, "loss": 0.0015, "step": 65 }, { "epoch": 3.6666666666666665, "grad_norm": 0.024880321696400642, "learning_rate": 4.343931245134616e-05, "loss": 0.0004, "step": 66 }, { "epoch": 3.7222222222222223, "grad_norm": 0.7327654361724854, "learning_rate": 4.028414082972141e-05, "loss": 0.0127, "step": 67 }, { "epoch": 3.7777777777777777, "grad_norm": 0.07404771447181702, "learning_rate": 3.721878753279017e-05, "loss": 0.0011, "step": 68 }, { "epoch": 3.8333333333333335, "grad_norm": 0.009151079691946507, "learning_rate": 3.424786314309365e-05, "loss": 0.0001, "step": 69 }, { "epoch": 3.888888888888889, "grad_norm": 0.03904305398464203, "learning_rate": 3.137583621312665e-05, "loss": 0.0004, "step": 70 }, { "epoch": 3.9444444444444446, "grad_norm": 0.16078825294971466, "learning_rate": 2.8607026544210114e-05, "loss": 0.0017, "step": 71 }, { "epoch": 4.0, "grad_norm": 0.035697516053915024, "learning_rate": 2.594559868909956e-05, "loss": 0.0005, "step": 72 }, { "epoch": 4.055555555555555, "grad_norm": 0.04049859568476677, "learning_rate": 2.339555568810221e-05, "loss": 0.0005, "step": 73 }, { "epoch": 4.111111111111111, "grad_norm": 0.04846888408064842, "learning_rate": 2.0960733048124083e-05, "loss": 0.0006, "step": 74 }, { "epoch": 4.166666666666667, "grad_norm": 0.015624168328940868, "learning_rate": 1.864479297370325e-05, "loss": 0.0002, "step": 75 }, { "epoch": 4.222222222222222, "grad_norm": 0.055838510394096375, "learning_rate": 1.6451218858706374e-05, "loss": 0.0008, "step": 76 }, { "epoch": 4.277777777777778, "grad_norm": 0.010197862051427364, "learning_rate": 1.4383310046973365e-05, "loss": 0.0001, "step": 77 }, { "epoch": 4.333333333333333, "grad_norm": 0.01613779552280903, "learning_rate": 1.2444176869790925e-05, "loss": 0.0003, "step": 78 }, { "epoch": 4.388888888888889, "grad_norm": 0.07378027588129044, "learning_rate": 1.0636735967658784e-05, "loss": 0.0006, "step": 79 }, { "epoch": 4.444444444444445, "grad_norm": 0.009738123044371605, "learning_rate": 8.963705903385345e-06, "loss": 0.0001, "step": 80 }, { "epoch": 4.5, "grad_norm": 0.024092555046081543, "learning_rate": 7.427603073110967e-06, "loss": 0.0002, "step": 81 }, { "epoch": 4.555555555555555, "grad_norm": 0.02920607477426529, "learning_rate": 6.030737921409169e-06, "loss": 0.0004, "step": 82 }, { "epoch": 4.611111111111111, "grad_norm": 0.033286549150943756, "learning_rate": 4.775211466158469e-06, "loss": 0.0005, "step": 83 }, { "epoch": 4.666666666666667, "grad_norm": 0.004122807644307613, "learning_rate": 3.662912138411967e-06, "loss": 0.0001, "step": 84 }, { "epoch": 4.722222222222222, "grad_norm": 0.016839250922203064, "learning_rate": 2.6955129420176196e-06, "loss": 0.0003, "step": 85 }, { "epoch": 4.777777777777778, "grad_norm": 0.013877292163670063, "learning_rate": 1.874468937261531e-06, "loss": 0.0002, "step": 86 }, { "epoch": 4.833333333333333, "grad_norm": 0.05045411363244057, "learning_rate": 1.201015052319099e-06, "loss": 0.0009, "step": 87 }, { "epoch": 4.888888888888889, "grad_norm": 0.050760798156261444, "learning_rate": 6.761642258056978e-07, "loss": 0.0005, "step": 88 }, { "epoch": 4.944444444444445, "grad_norm": 0.08366449922323227, "learning_rate": 3.007058832207976e-07, "loss": 0.0012, "step": 89 }, { "epoch": 5.0, "grad_norm": 0.15823712944984436, "learning_rate": 7.520474957699586e-08, "loss": 0.0017, "step": 90 } ], "logging_steps": 1, "max_steps": 90, "num_input_tokens_seen": 0, "num_train_epochs": 5, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 1.7422394887411668e+16, "train_batch_size": 4, "trial_name": null, "trial_params": null }