| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 2.5579014715291106, |
| "eval_steps": 200, |
| "global_step": 1000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.025591810620601407, |
| "grad_norm": 78.5, |
| "learning_rate": 1.35e-07, |
| "loss": 2.687112236022949, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.05118362124120281, |
| "grad_norm": 117.0, |
| "learning_rate": 2.85e-07, |
| "loss": 2.8489315032958986, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.07677543186180422, |
| "grad_norm": 302.0, |
| "learning_rate": 4.3499999999999996e-07, |
| "loss": 2.7195245742797853, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.10236724248240563, |
| "grad_norm": 328.0, |
| "learning_rate": 5.85e-07, |
| "loss": 2.9164562225341797, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.12795905310300704, |
| "grad_norm": 109.5, |
| "learning_rate": 7.350000000000001e-07, |
| "loss": 2.7758764266967773, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.15355086372360843, |
| "grad_norm": 137.0, |
| "learning_rate": 8.85e-07, |
| "loss": 2.842637825012207, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.17914267434420986, |
| "grad_norm": 230.0, |
| "learning_rate": 1.035e-06, |
| "loss": 2.8522205352783203, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.20473448496481125, |
| "grad_norm": 183.0, |
| "learning_rate": 1.185e-06, |
| "loss": 2.8120689392089844, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.23032629558541268, |
| "grad_norm": 178.0, |
| "learning_rate": 1.3350000000000001e-06, |
| "loss": 2.547420883178711, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.2559181062060141, |
| "grad_norm": 57.25, |
| "learning_rate": 1.485e-06, |
| "loss": 2.685759735107422, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.28150991682661547, |
| "grad_norm": 162.0, |
| "learning_rate": 1.6350000000000002e-06, |
| "loss": 2.9074512481689454, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.30710172744721687, |
| "grad_norm": 133.0, |
| "learning_rate": 1.785e-06, |
| "loss": 2.6075531005859376, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.3326935380678183, |
| "grad_norm": 76.0, |
| "learning_rate": 1.935e-06, |
| "loss": 2.26098575592041, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.3582853486884197, |
| "grad_norm": 95.0, |
| "learning_rate": 2.085e-06, |
| "loss": 2.113755226135254, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.3838771593090211, |
| "grad_norm": 96.5, |
| "learning_rate": 2.235e-06, |
| "loss": 2.3181507110595705, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.4094689699296225, |
| "grad_norm": 122.0, |
| "learning_rate": 2.385e-06, |
| "loss": 2.0230716705322265, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.4350607805502239, |
| "grad_norm": 167.0, |
| "learning_rate": 2.535e-06, |
| "loss": 2.166606140136719, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.46065259117082535, |
| "grad_norm": 103.0, |
| "learning_rate": 2.685e-06, |
| "loss": 2.094511795043945, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.48624440179142675, |
| "grad_norm": 97.5, |
| "learning_rate": 2.835e-06, |
| "loss": 1.9705234527587892, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.5118362124120281, |
| "grad_norm": 136.0, |
| "learning_rate": 2.9850000000000002e-06, |
| "loss": 1.8970783233642579, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.5118362124120281, |
| "eval_loss": 0.3652507960796356, |
| "eval_runtime": 20.6147, |
| "eval_samples_per_second": 97.018, |
| "eval_steps_per_second": 3.056, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.5374280230326296, |
| "grad_norm": 75.0, |
| "learning_rate": 2.9850000000000002e-06, |
| "loss": 1.8157735824584962, |
| "step": 210 |
| }, |
| { |
| "epoch": 0.5630198336532309, |
| "grad_norm": 170.0, |
| "learning_rate": 2.9683333333333334e-06, |
| "loss": 1.7747406005859374, |
| "step": 220 |
| }, |
| { |
| "epoch": 0.5886116442738324, |
| "grad_norm": 117.5, |
| "learning_rate": 2.9516666666666666e-06, |
| "loss": 1.8584400177001954, |
| "step": 230 |
| }, |
| { |
| "epoch": 0.6142034548944337, |
| "grad_norm": 55.5, |
| "learning_rate": 2.9350000000000003e-06, |
| "loss": 1.7559932708740233, |
| "step": 240 |
| }, |
| { |
| "epoch": 0.6397952655150352, |
| "grad_norm": 111.5, |
| "learning_rate": 2.9183333333333335e-06, |
| "loss": 1.938743782043457, |
| "step": 250 |
| }, |
| { |
| "epoch": 0.6653870761356366, |
| "grad_norm": 153.0, |
| "learning_rate": 2.9016666666666667e-06, |
| "loss": 1.6487155914306642, |
| "step": 260 |
| }, |
| { |
| "epoch": 0.690978886756238, |
| "grad_norm": 83.0, |
| "learning_rate": 2.885e-06, |
| "loss": 1.5814908027648926, |
| "step": 270 |
| }, |
| { |
| "epoch": 0.7165706973768394, |
| "grad_norm": 50.75, |
| "learning_rate": 2.8683333333333335e-06, |
| "loss": 1.5237090110778808, |
| "step": 280 |
| }, |
| { |
| "epoch": 0.7421625079974408, |
| "grad_norm": 79.0, |
| "learning_rate": 2.8516666666666668e-06, |
| "loss": 1.5812367439270019, |
| "step": 290 |
| }, |
| { |
| "epoch": 0.7677543186180422, |
| "grad_norm": 61.25, |
| "learning_rate": 2.835e-06, |
| "loss": 1.499494743347168, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.7933461292386437, |
| "grad_norm": 104.5, |
| "learning_rate": 2.818333333333333e-06, |
| "loss": 1.393200969696045, |
| "step": 310 |
| }, |
| { |
| "epoch": 0.818937939859245, |
| "grad_norm": 101.5, |
| "learning_rate": 2.801666666666667e-06, |
| "loss": 1.5071249961853028, |
| "step": 320 |
| }, |
| { |
| "epoch": 0.8445297504798465, |
| "grad_norm": 109.5, |
| "learning_rate": 2.785e-06, |
| "loss": 1.3955337524414062, |
| "step": 330 |
| }, |
| { |
| "epoch": 0.8701215611004478, |
| "grad_norm": 86.0, |
| "learning_rate": 2.7683333333333337e-06, |
| "loss": 1.1970922470092773, |
| "step": 340 |
| }, |
| { |
| "epoch": 0.8957133717210493, |
| "grad_norm": 127.0, |
| "learning_rate": 2.751666666666667e-06, |
| "loss": 1.4966418266296386, |
| "step": 350 |
| }, |
| { |
| "epoch": 0.9213051823416507, |
| "grad_norm": 60.0, |
| "learning_rate": 2.735e-06, |
| "loss": 1.5355607986450195, |
| "step": 360 |
| }, |
| { |
| "epoch": 0.946896992962252, |
| "grad_norm": 110.0, |
| "learning_rate": 2.7183333333333333e-06, |
| "loss": 1.2876726150512696, |
| "step": 370 |
| }, |
| { |
| "epoch": 0.9724888035828535, |
| "grad_norm": 61.75, |
| "learning_rate": 2.701666666666667e-06, |
| "loss": 1.4528013229370118, |
| "step": 380 |
| }, |
| { |
| "epoch": 0.9980806142034548, |
| "grad_norm": 73.0, |
| "learning_rate": 2.685e-06, |
| "loss": 1.4765524864196777, |
| "step": 390 |
| }, |
| { |
| "epoch": 1.0230326295585412, |
| "grad_norm": 43.5, |
| "learning_rate": 2.6683333333333333e-06, |
| "loss": 1.31887845993042, |
| "step": 400 |
| }, |
| { |
| "epoch": 1.0230326295585412, |
| "eval_loss": 0.25503993034362793, |
| "eval_runtime": 20.8252, |
| "eval_samples_per_second": 96.037, |
| "eval_steps_per_second": 3.025, |
| "step": 400 |
| }, |
| { |
| "epoch": 1.0486244401791427, |
| "grad_norm": 116.5, |
| "learning_rate": 2.6516666666666665e-06, |
| "loss": 1.2159116744995118, |
| "step": 410 |
| }, |
| { |
| "epoch": 1.0742162507997441, |
| "grad_norm": 124.5, |
| "learning_rate": 2.6349999999999998e-06, |
| "loss": 1.346174716949463, |
| "step": 420 |
| }, |
| { |
| "epoch": 1.0998080614203456, |
| "grad_norm": 70.0, |
| "learning_rate": 2.6183333333333334e-06, |
| "loss": 1.3012319564819337, |
| "step": 430 |
| }, |
| { |
| "epoch": 1.1253998720409468, |
| "grad_norm": 78.5, |
| "learning_rate": 2.6016666666666666e-06, |
| "loss": 1.2842123985290528, |
| "step": 440 |
| }, |
| { |
| "epoch": 1.1509916826615483, |
| "grad_norm": 95.5, |
| "learning_rate": 2.5850000000000002e-06, |
| "loss": 1.1591164588928222, |
| "step": 450 |
| }, |
| { |
| "epoch": 1.1765834932821497, |
| "grad_norm": 119.5, |
| "learning_rate": 2.5683333333333334e-06, |
| "loss": 1.28289794921875, |
| "step": 460 |
| }, |
| { |
| "epoch": 1.2021753039027512, |
| "grad_norm": 47.25, |
| "learning_rate": 2.5516666666666667e-06, |
| "loss": 1.2217205047607422, |
| "step": 470 |
| }, |
| { |
| "epoch": 1.2277671145233526, |
| "grad_norm": 61.0, |
| "learning_rate": 2.535e-06, |
| "loss": 1.3094758987426758, |
| "step": 480 |
| }, |
| { |
| "epoch": 1.2533589251439539, |
| "grad_norm": 125.5, |
| "learning_rate": 2.5183333333333335e-06, |
| "loss": 1.0260610580444336, |
| "step": 490 |
| }, |
| { |
| "epoch": 1.2789507357645553, |
| "grad_norm": 52.75, |
| "learning_rate": 2.5016666666666667e-06, |
| "loss": 1.3739563941955566, |
| "step": 500 |
| }, |
| { |
| "epoch": 1.3045425463851568, |
| "grad_norm": 101.5, |
| "learning_rate": 2.4850000000000003e-06, |
| "loss": 1.1762229919433593, |
| "step": 510 |
| }, |
| { |
| "epoch": 1.3301343570057582, |
| "grad_norm": 74.0, |
| "learning_rate": 2.4683333333333336e-06, |
| "loss": 1.2967172622680665, |
| "step": 520 |
| }, |
| { |
| "epoch": 1.3557261676263597, |
| "grad_norm": 57.25, |
| "learning_rate": 2.4516666666666668e-06, |
| "loss": 1.260681438446045, |
| "step": 530 |
| }, |
| { |
| "epoch": 1.381317978246961, |
| "grad_norm": 88.0, |
| "learning_rate": 2.435e-06, |
| "loss": 1.0191940307617187, |
| "step": 540 |
| }, |
| { |
| "epoch": 1.4069097888675623, |
| "grad_norm": 129.0, |
| "learning_rate": 2.418333333333333e-06, |
| "loss": 1.0855172157287598, |
| "step": 550 |
| }, |
| { |
| "epoch": 1.4325015994881638, |
| "grad_norm": 74.0, |
| "learning_rate": 2.401666666666667e-06, |
| "loss": 1.1898975372314453, |
| "step": 560 |
| }, |
| { |
| "epoch": 1.4580934101087653, |
| "grad_norm": 54.25, |
| "learning_rate": 2.385e-06, |
| "loss": 1.1281633377075195, |
| "step": 570 |
| }, |
| { |
| "epoch": 1.4836852207293667, |
| "grad_norm": 86.5, |
| "learning_rate": 2.3683333333333332e-06, |
| "loss": 1.4006970405578614, |
| "step": 580 |
| }, |
| { |
| "epoch": 1.509277031349968, |
| "grad_norm": 96.5, |
| "learning_rate": 2.3516666666666665e-06, |
| "loss": 1.1684344291687012, |
| "step": 590 |
| }, |
| { |
| "epoch": 1.5348688419705694, |
| "grad_norm": 110.0, |
| "learning_rate": 2.335e-06, |
| "loss": 1.2921236038208008, |
| "step": 600 |
| }, |
| { |
| "epoch": 1.5348688419705694, |
| "eval_loss": 0.2259225696325302, |
| "eval_runtime": 21.0241, |
| "eval_samples_per_second": 95.129, |
| "eval_steps_per_second": 2.997, |
| "step": 600 |
| }, |
| { |
| "epoch": 1.5604606525911708, |
| "grad_norm": 88.5, |
| "learning_rate": 2.3183333333333333e-06, |
| "loss": 1.076922607421875, |
| "step": 610 |
| }, |
| { |
| "epoch": 1.586052463211772, |
| "grad_norm": 105.0, |
| "learning_rate": 2.301666666666667e-06, |
| "loss": 1.0854657173156739, |
| "step": 620 |
| }, |
| { |
| "epoch": 1.6116442738323737, |
| "grad_norm": 70.5, |
| "learning_rate": 2.285e-06, |
| "loss": 1.244626522064209, |
| "step": 630 |
| }, |
| { |
| "epoch": 1.637236084452975, |
| "grad_norm": 30.0, |
| "learning_rate": 2.2683333333333334e-06, |
| "loss": 1.047581386566162, |
| "step": 640 |
| }, |
| { |
| "epoch": 1.6628278950735764, |
| "grad_norm": 59.75, |
| "learning_rate": 2.2516666666666666e-06, |
| "loss": 1.1024110794067383, |
| "step": 650 |
| }, |
| { |
| "epoch": 1.6884197056941779, |
| "grad_norm": 67.0, |
| "learning_rate": 2.235e-06, |
| "loss": 1.0247148513793944, |
| "step": 660 |
| }, |
| { |
| "epoch": 1.714011516314779, |
| "grad_norm": 66.0, |
| "learning_rate": 2.2183333333333334e-06, |
| "loss": 1.162140464782715, |
| "step": 670 |
| }, |
| { |
| "epoch": 1.7396033269353808, |
| "grad_norm": 79.5, |
| "learning_rate": 2.2016666666666666e-06, |
| "loss": 1.058082389831543, |
| "step": 680 |
| }, |
| { |
| "epoch": 1.765195137555982, |
| "grad_norm": 103.5, |
| "learning_rate": 2.1850000000000003e-06, |
| "loss": 1.0551362037658691, |
| "step": 690 |
| }, |
| { |
| "epoch": 1.7907869481765835, |
| "grad_norm": 41.5, |
| "learning_rate": 2.1683333333333335e-06, |
| "loss": 1.0396892547607421, |
| "step": 700 |
| }, |
| { |
| "epoch": 1.816378758797185, |
| "grad_norm": 84.0, |
| "learning_rate": 2.1516666666666667e-06, |
| "loss": 1.0307888984680176, |
| "step": 710 |
| }, |
| { |
| "epoch": 1.8419705694177864, |
| "grad_norm": 13.0625, |
| "learning_rate": 2.135e-06, |
| "loss": 1.1348337173461913, |
| "step": 720 |
| }, |
| { |
| "epoch": 1.8675623800383878, |
| "grad_norm": 60.5, |
| "learning_rate": 2.1183333333333335e-06, |
| "loss": 1.1178834915161133, |
| "step": 730 |
| }, |
| { |
| "epoch": 1.893154190658989, |
| "grad_norm": 90.5, |
| "learning_rate": 2.1016666666666667e-06, |
| "loss": 0.8540509223937989, |
| "step": 740 |
| }, |
| { |
| "epoch": 1.9187460012795905, |
| "grad_norm": 68.0, |
| "learning_rate": 2.085e-06, |
| "loss": 1.0177457809448243, |
| "step": 750 |
| }, |
| { |
| "epoch": 1.944337811900192, |
| "grad_norm": 56.75, |
| "learning_rate": 2.068333333333333e-06, |
| "loss": 0.9258890151977539, |
| "step": 760 |
| }, |
| { |
| "epoch": 1.9699296225207934, |
| "grad_norm": 105.5, |
| "learning_rate": 2.0516666666666668e-06, |
| "loss": 1.1195788383483887, |
| "step": 770 |
| }, |
| { |
| "epoch": 1.9955214331413949, |
| "grad_norm": 65.5, |
| "learning_rate": 2.035e-06, |
| "loss": 1.0428399085998534, |
| "step": 780 |
| }, |
| { |
| "epoch": 2.0204734484964813, |
| "grad_norm": 199.0, |
| "learning_rate": 2.0183333333333336e-06, |
| "loss": 1.0544426918029786, |
| "step": 790 |
| }, |
| { |
| "epoch": 2.0460652591170825, |
| "grad_norm": 62.0, |
| "learning_rate": 2.001666666666667e-06, |
| "loss": 0.9359260559082031, |
| "step": 800 |
| }, |
| { |
| "epoch": 2.0460652591170825, |
| "eval_loss": 0.20818977057933807, |
| "eval_runtime": 20.9478, |
| "eval_samples_per_second": 95.475, |
| "eval_steps_per_second": 3.007, |
| "step": 800 |
| }, |
| { |
| "epoch": 2.071657069737684, |
| "grad_norm": 78.0, |
| "learning_rate": 1.985e-06, |
| "loss": 1.1695940017700195, |
| "step": 810 |
| }, |
| { |
| "epoch": 2.0972488803582854, |
| "grad_norm": 51.75, |
| "learning_rate": 1.9683333333333333e-06, |
| "loss": 1.0586297988891602, |
| "step": 820 |
| }, |
| { |
| "epoch": 2.1228406909788866, |
| "grad_norm": 63.25, |
| "learning_rate": 1.951666666666667e-06, |
| "loss": 0.9728346824645996, |
| "step": 830 |
| }, |
| { |
| "epoch": 2.1484325015994883, |
| "grad_norm": 97.0, |
| "learning_rate": 1.935e-06, |
| "loss": 1.2120585441589355, |
| "step": 840 |
| }, |
| { |
| "epoch": 2.1740243122200895, |
| "grad_norm": 78.5, |
| "learning_rate": 1.9183333333333333e-06, |
| "loss": 1.0046463012695312, |
| "step": 850 |
| }, |
| { |
| "epoch": 2.199616122840691, |
| "grad_norm": 82.5, |
| "learning_rate": 1.9016666666666665e-06, |
| "loss": 1.0721829414367676, |
| "step": 860 |
| }, |
| { |
| "epoch": 2.2252079334612924, |
| "grad_norm": 88.0, |
| "learning_rate": 1.885e-06, |
| "loss": 0.8954225540161133, |
| "step": 870 |
| }, |
| { |
| "epoch": 2.2507997440818936, |
| "grad_norm": 56.25, |
| "learning_rate": 1.8683333333333334e-06, |
| "loss": 1.1441267013549805, |
| "step": 880 |
| }, |
| { |
| "epoch": 2.2763915547024953, |
| "grad_norm": 86.5, |
| "learning_rate": 1.8516666666666668e-06, |
| "loss": 0.9673049926757813, |
| "step": 890 |
| }, |
| { |
| "epoch": 2.3019833653230966, |
| "grad_norm": 107.5, |
| "learning_rate": 1.8350000000000002e-06, |
| "loss": 0.9580348014831543, |
| "step": 900 |
| }, |
| { |
| "epoch": 2.3275751759436982, |
| "grad_norm": 47.75, |
| "learning_rate": 1.8183333333333334e-06, |
| "loss": 1.026679801940918, |
| "step": 910 |
| }, |
| { |
| "epoch": 2.3531669865642995, |
| "grad_norm": 54.5, |
| "learning_rate": 1.8016666666666666e-06, |
| "loss": 0.9906099319458008, |
| "step": 920 |
| }, |
| { |
| "epoch": 2.3787587971849007, |
| "grad_norm": 64.5, |
| "learning_rate": 1.785e-06, |
| "loss": 1.017255687713623, |
| "step": 930 |
| }, |
| { |
| "epoch": 2.4043506078055024, |
| "grad_norm": 47.75, |
| "learning_rate": 1.7683333333333333e-06, |
| "loss": 0.8187576293945312, |
| "step": 940 |
| }, |
| { |
| "epoch": 2.4299424184261036, |
| "grad_norm": 116.5, |
| "learning_rate": 1.7516666666666667e-06, |
| "loss": 1.147125816345215, |
| "step": 950 |
| }, |
| { |
| "epoch": 2.4555342290467053, |
| "grad_norm": 75.5, |
| "learning_rate": 1.7350000000000001e-06, |
| "loss": 1.1997849464416503, |
| "step": 960 |
| }, |
| { |
| "epoch": 2.4811260396673065, |
| "grad_norm": 42.0, |
| "learning_rate": 1.7183333333333335e-06, |
| "loss": 0.8572607040405273, |
| "step": 970 |
| }, |
| { |
| "epoch": 2.5067178502879077, |
| "grad_norm": 120.0, |
| "learning_rate": 1.7016666666666665e-06, |
| "loss": 0.9942266464233398, |
| "step": 980 |
| }, |
| { |
| "epoch": 2.5323096609085094, |
| "grad_norm": 81.0, |
| "learning_rate": 1.685e-06, |
| "loss": 0.9823317527770996, |
| "step": 990 |
| }, |
| { |
| "epoch": 2.5579014715291106, |
| "grad_norm": 34.5, |
| "learning_rate": 1.6683333333333334e-06, |
| "loss": 1.0651761054992677, |
| "step": 1000 |
| }, |
| { |
| "epoch": 2.5579014715291106, |
| "eval_loss": 0.20452462136745453, |
| "eval_runtime": 20.9772, |
| "eval_samples_per_second": 95.341, |
| "eval_steps_per_second": 3.003, |
| "step": 1000 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 2000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 6, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 2.3839710759498547e+17, |
| "train_batch_size": 32, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|