| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.2, |
| "eval_steps": 1000, |
| "global_step": 10000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 0.8063191175460815, |
| "learning_rate": 7.920000000000001e-05, |
| "loss": 8.2537, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 0.6161019206047058, |
| "learning_rate": 0.00015920000000000002, |
| "loss": 6.5731, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 0.8347169160842896, |
| "learning_rate": 0.00023920000000000001, |
| "loss": 6.1639, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.9150793552398682, |
| "learning_rate": 0.0003192, |
| "loss": 5.8682, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 0.6698479652404785, |
| "learning_rate": 0.0003992, |
| "loss": 5.5708, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.5323518514633179, |
| "learning_rate": 0.00047920000000000005, |
| "loss": 5.3157, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 0.7472386956214905, |
| "learning_rate": 0.0005592, |
| "loss": 5.0907, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.6033297181129456, |
| "learning_rate": 0.0006392, |
| "loss": 4.8805, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 0.497113436460495, |
| "learning_rate": 0.0007191999999999999, |
| "loss": 4.7207, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.47698545455932617, |
| "learning_rate": 0.0007992, |
| "loss": 4.6068, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_accuracy": 0.2703913894324853, |
| "eval_loss": 4.5053935050964355, |
| "eval_runtime": 13.0992, |
| "eval_samples_per_second": 76.341, |
| "eval_steps_per_second": 0.611, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 0.44341349601745605, |
| "learning_rate": 0.0008792, |
| "loss": 4.4904, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.4383724331855774, |
| "learning_rate": 0.0009592000000000001, |
| "loss": 4.439, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 0.3894674777984619, |
| "learning_rate": 0.0010391999999999999, |
| "loss": 4.3735, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.41903820633888245, |
| "learning_rate": 0.0011192, |
| "loss": 4.3406, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 0.36601418256759644, |
| "learning_rate": 0.0011992, |
| "loss": 4.2879, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.4282030165195465, |
| "learning_rate": 0.0012791999999999999, |
| "loss": 4.2625, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 0.3431214690208435, |
| "learning_rate": 0.0013592, |
| "loss": 4.2316, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.3303460478782654, |
| "learning_rate": 0.0014392, |
| "loss": 4.1976, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 0.3030431866645813, |
| "learning_rate": 0.0015192, |
| "loss": 4.1494, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.3076256513595581, |
| "learning_rate": 0.0015992, |
| "loss": 4.1688, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.30658512720156555, |
| "eval_loss": 4.094621658325195, |
| "eval_runtime": 12.4723, |
| "eval_samples_per_second": 80.178, |
| "eval_steps_per_second": 0.641, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 0.2820931077003479, |
| "learning_rate": 0.0016792, |
| "loss": 4.5201, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.23724448680877686, |
| "learning_rate": 0.0017592, |
| "loss": 4.1333, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 0.2325439602136612, |
| "learning_rate": 0.0018392, |
| "loss": 4.1072, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.23736131191253662, |
| "learning_rate": 0.0019192, |
| "loss": 4.0682, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.2313733696937561, |
| "learning_rate": 0.0019992, |
| "loss": 4.0571, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.22318924963474274, |
| "learning_rate": 0.001999978563623903, |
| "loss": 4.0153, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 0.20275355875492096, |
| "learning_rate": 0.0019999133871331223, |
| "loss": 4.0163, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.21306632459163666, |
| "learning_rate": 0.0019998044711915177, |
| "loss": 3.9777, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 0.19690750539302826, |
| "learning_rate": 0.0019996518205634257, |
| "loss": 3.9603, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.20464760065078735, |
| "learning_rate": 0.0019994554419262797, |
| "loss": 3.9566, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.3226555772994129, |
| "eval_loss": 3.9112932682037354, |
| "eval_runtime": 11.7171, |
| "eval_samples_per_second": 85.345, |
| "eval_steps_per_second": 0.683, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 0.18151508271694183, |
| "learning_rate": 0.001999215343870317, |
| "loss": 3.9501, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 0.18748600780963898, |
| "learning_rate": 0.0019989315368982054, |
| "loss": 3.9228, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 0.17895972728729248, |
| "learning_rate": 0.0019986040334245797, |
| "loss": 3.9041, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 0.18976759910583496, |
| "learning_rate": 0.001998232847775504, |
| "loss": 3.9123, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 0.1871727555990219, |
| "learning_rate": 0.0019978179961878404, |
| "loss": 3.8846, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 0.17986708879470825, |
| "learning_rate": 0.001997359496808541, |
| "loss": 3.8794, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 0.17090782523155212, |
| "learning_rate": 0.001996857369693855, |
| "loss": 3.8593, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 0.19015273451805115, |
| "learning_rate": 0.0019963116368084486, |
| "loss": 3.8642, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 0.1871633529663086, |
| "learning_rate": 0.001995722322024446, |
| "loss": 3.8479, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 0.18049727380275726, |
| "learning_rate": 0.001995089451120385, |
| "loss": 3.8226, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.33194520547945205, |
| "eval_loss": 3.8037452697753906, |
| "eval_runtime": 12.6853, |
| "eval_samples_per_second": 78.832, |
| "eval_steps_per_second": 0.631, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.082, |
| "grad_norm": 0.14800746738910675, |
| "learning_rate": 0.0019944130517800893, |
| "loss": 3.8319, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.084, |
| "grad_norm": 0.14713813364505768, |
| "learning_rate": 0.001993693153591457, |
| "loss": 3.8319, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.086, |
| "grad_norm": 0.17019635438919067, |
| "learning_rate": 0.0019929297880451674, |
| "loss": 3.8095, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.088, |
| "grad_norm": 0.15243591368198395, |
| "learning_rate": 0.0019921229885333023, |
| "loss": 3.8003, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.09, |
| "grad_norm": 0.16134986281394958, |
| "learning_rate": 0.001991272790347886, |
| "loss": 3.814, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.092, |
| "grad_norm": 0.16883811354637146, |
| "learning_rate": 0.0019903792306793415, |
| "loss": 3.7822, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.094, |
| "grad_norm": 0.15718473494052887, |
| "learning_rate": 0.001989442348614863, |
| "loss": 3.7902, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.096, |
| "grad_norm": 0.14631035923957825, |
| "learning_rate": 0.0019884621851367075, |
| "loss": 3.7805, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.098, |
| "grad_norm": 0.14771369099617004, |
| "learning_rate": 0.001987438783120401, |
| "loss": 3.7778, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 0.1536298543214798, |
| "learning_rate": 0.001986372187332862, |
| "loss": 3.7848, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.1, |
| "eval_accuracy": 0.340252446183953, |
| "eval_loss": 3.7290549278259277, |
| "eval_runtime": 11.9375, |
| "eval_samples_per_second": 83.77, |
| "eval_steps_per_second": 0.67, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.102, |
| "grad_norm": 0.15110976994037628, |
| "learning_rate": 0.0019852624444304467, |
| "loss": 3.7502, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.104, |
| "grad_norm": 0.15862922370433807, |
| "learning_rate": 0.0019841096029569044, |
| "loss": 3.761, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.106, |
| "grad_norm": 0.13948658108711243, |
| "learning_rate": 0.0019829137133412556, |
| "loss": 3.7591, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.108, |
| "grad_norm": 0.148148313164711, |
| "learning_rate": 0.001981674827895587, |
| "loss": 3.7385, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.11, |
| "grad_norm": 0.13170774281024933, |
| "learning_rate": 0.00198039300081276, |
| "loss": 3.7424, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.112, |
| "grad_norm": 0.14352479577064514, |
| "learning_rate": 0.0019790682881640448, |
| "loss": 3.7423, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.114, |
| "grad_norm": 0.12865835428237915, |
| "learning_rate": 0.001977700747896664, |
| "loss": 3.7263, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.116, |
| "grad_norm": 0.15470698475837708, |
| "learning_rate": 0.001976290439831259, |
| "loss": 3.7297, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.118, |
| "grad_norm": 0.1353532075881958, |
| "learning_rate": 0.0019748374256592736, |
| "loss": 3.7218, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 0.14982788264751434, |
| "learning_rate": 0.0019733417689402543, |
| "loss": 3.7164, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.12, |
| "eval_accuracy": 0.34496673189823873, |
| "eval_loss": 3.6746184825897217, |
| "eval_runtime": 11.9904, |
| "eval_samples_per_second": 83.4, |
| "eval_steps_per_second": 0.667, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.122, |
| "grad_norm": 0.14931881427764893, |
| "learning_rate": 0.0019718035350990712, |
| "loss": 3.7169, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.124, |
| "grad_norm": 0.14251381158828735, |
| "learning_rate": 0.0019702227914230566, |
| "loss": 3.6929, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.126, |
| "grad_norm": 0.14251849055290222, |
| "learning_rate": 0.001968599607059059, |
| "loss": 3.7018, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 0.15696577727794647, |
| "learning_rate": 0.0019669340530104207, |
| "loss": 3.7034, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.13, |
| "grad_norm": 0.14737001061439514, |
| "learning_rate": 0.001965226202133872, |
| "loss": 3.6889, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.132, |
| "grad_norm": 0.13667502999305725, |
| "learning_rate": 0.0019634761291363427, |
| "loss": 3.6935, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.134, |
| "grad_norm": 0.13390734791755676, |
| "learning_rate": 0.0019616839105716954, |
| "loss": 3.7032, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.136, |
| "grad_norm": 0.14834214746952057, |
| "learning_rate": 0.0019598496248373755, |
| "loss": 3.6657, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.138, |
| "grad_norm": 0.1314767599105835, |
| "learning_rate": 0.001957973352170984, |
| "loss": 3.673, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 0.15161661803722382, |
| "learning_rate": 0.001956055174646765, |
| "loss": 3.6713, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.14, |
| "eval_accuracy": 0.3484422700587084, |
| "eval_loss": 3.6349546909332275, |
| "eval_runtime": 12.1265, |
| "eval_samples_per_second": 82.464, |
| "eval_steps_per_second": 0.66, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.142, |
| "grad_norm": 0.12710200250148773, |
| "learning_rate": 0.0019540951761720174, |
| "loss": 3.67, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.144, |
| "grad_norm": 0.18498213589191437, |
| "learning_rate": 0.0019520934424834247, |
| "loss": 3.675, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.146, |
| "grad_norm": 0.13697299361228943, |
| "learning_rate": 0.0019500500611433025, |
| "loss": 3.6497, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.148, |
| "grad_norm": 0.13290269672870636, |
| "learning_rate": 0.0019479651215357707, |
| "loss": 3.662, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 0.12744054198265076, |
| "learning_rate": 0.0019458387148628417, |
| "loss": 3.6645, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.152, |
| "grad_norm": 0.13697493076324463, |
| "learning_rate": 0.001943670934140432, |
| "loss": 3.6374, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.154, |
| "grad_norm": 0.14067181944847107, |
| "learning_rate": 0.0019414618741942936, |
| "loss": 3.6453, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.156, |
| "grad_norm": 0.1316055953502655, |
| "learning_rate": 0.0019392116316558638, |
| "loss": 3.6679, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.158, |
| "grad_norm": 0.1261526495218277, |
| "learning_rate": 0.001936920304958042, |
| "loss": 3.6343, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 0.13644501566886902, |
| "learning_rate": 0.0019345879943308804, |
| "loss": 3.6365, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.16, |
| "eval_accuracy": 0.35229158512720155, |
| "eval_loss": 3.5980424880981445, |
| "eval_runtime": 17.6098, |
| "eval_samples_per_second": 56.787, |
| "eval_steps_per_second": 0.454, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.162, |
| "grad_norm": 0.13911907374858856, |
| "learning_rate": 0.0019322148017972016, |
| "loss": 3.646, |
| "step": 8100 |
| }, |
| { |
| "epoch": 0.164, |
| "grad_norm": 0.1297680288553238, |
| "learning_rate": 0.001929800831168135, |
| "loss": 3.6412, |
| "step": 8200 |
| }, |
| { |
| "epoch": 0.166, |
| "grad_norm": 0.14803367853164673, |
| "learning_rate": 0.001927346188038576, |
| "loss": 3.6298, |
| "step": 8300 |
| }, |
| { |
| "epoch": 0.168, |
| "grad_norm": 0.1325535923242569, |
| "learning_rate": 0.0019248509797825672, |
| "loss": 3.6067, |
| "step": 8400 |
| }, |
| { |
| "epoch": 0.17, |
| "grad_norm": 0.16948860883712769, |
| "learning_rate": 0.0019223153155486009, |
| "loss": 3.6284, |
| "step": 8500 |
| }, |
| { |
| "epoch": 0.172, |
| "grad_norm": 0.12973898649215698, |
| "learning_rate": 0.0019197393062548454, |
| "loss": 3.627, |
| "step": 8600 |
| }, |
| { |
| "epoch": 0.174, |
| "grad_norm": 0.1576964110136032, |
| "learning_rate": 0.0019171230645842923, |
| "loss": 3.6027, |
| "step": 8700 |
| }, |
| { |
| "epoch": 0.176, |
| "grad_norm": 0.15394999086856842, |
| "learning_rate": 0.0019144667049798272, |
| "loss": 3.6098, |
| "step": 8800 |
| }, |
| { |
| "epoch": 0.178, |
| "grad_norm": 0.12594962120056152, |
| "learning_rate": 0.0019117703436392253, |
| "loss": 3.6221, |
| "step": 8900 |
| }, |
| { |
| "epoch": 0.18, |
| "grad_norm": 0.1420104056596756, |
| "learning_rate": 0.001909034098510066, |
| "loss": 3.5993, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.18, |
| "eval_accuracy": 0.3549628180039139, |
| "eval_loss": 3.570964813232422, |
| "eval_runtime": 12.4303, |
| "eval_samples_per_second": 80.449, |
| "eval_steps_per_second": 0.644, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.182, |
| "grad_norm": 0.12720872461795807, |
| "learning_rate": 0.001906258089284576, |
| "loss": 3.5913, |
| "step": 9100 |
| }, |
| { |
| "epoch": 0.184, |
| "grad_norm": 0.14890190958976746, |
| "learning_rate": 0.0019034424373943915, |
| "loss": 3.6113, |
| "step": 9200 |
| }, |
| { |
| "epoch": 0.186, |
| "grad_norm": 0.15023267269134521, |
| "learning_rate": 0.0019005872660052478, |
| "loss": 3.6011, |
| "step": 9300 |
| }, |
| { |
| "epoch": 0.188, |
| "grad_norm": 0.1637164056301117, |
| "learning_rate": 0.001897692700011591, |
| "loss": 3.5932, |
| "step": 9400 |
| }, |
| { |
| "epoch": 0.19, |
| "grad_norm": 0.13102982938289642, |
| "learning_rate": 0.0018947588660311143, |
| "loss": 3.5889, |
| "step": 9500 |
| }, |
| { |
| "epoch": 0.192, |
| "grad_norm": 0.16267850995063782, |
| "learning_rate": 0.0018917858923992211, |
| "loss": 3.5939, |
| "step": 9600 |
| }, |
| { |
| "epoch": 0.194, |
| "grad_norm": 0.12798982858657837, |
| "learning_rate": 0.0018887739091634085, |
| "loss": 3.5896, |
| "step": 9700 |
| }, |
| { |
| "epoch": 0.196, |
| "grad_norm": 0.13872383534908295, |
| "learning_rate": 0.0018857230480775807, |
| "loss": 3.574, |
| "step": 9800 |
| }, |
| { |
| "epoch": 0.198, |
| "grad_norm": 0.11993825435638428, |
| "learning_rate": 0.0018826334425962855, |
| "loss": 3.589, |
| "step": 9900 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 0.13074541091918945, |
| "learning_rate": 0.0018795052278688753, |
| "loss": 3.5828, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.2, |
| "eval_accuracy": 0.35816046966731896, |
| "eval_loss": 3.540745973587036, |
| "eval_runtime": 17.8387, |
| "eval_samples_per_second": 56.058, |
| "eval_steps_per_second": 0.448, |
| "step": 10000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 128, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|