| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.6, |
| "eval_steps": 1000, |
| "global_step": 30000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 0.8063191175460815, |
| "learning_rate": 7.920000000000001e-05, |
| "loss": 8.2537, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 0.6161019206047058, |
| "learning_rate": 0.00015920000000000002, |
| "loss": 6.5731, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 0.8347169160842896, |
| "learning_rate": 0.00023920000000000001, |
| "loss": 6.1639, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.9150793552398682, |
| "learning_rate": 0.0003192, |
| "loss": 5.8682, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 0.6698479652404785, |
| "learning_rate": 0.0003992, |
| "loss": 5.5708, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.5323518514633179, |
| "learning_rate": 0.00047920000000000005, |
| "loss": 5.3157, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 0.7472386956214905, |
| "learning_rate": 0.0005592, |
| "loss": 5.0907, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.6033297181129456, |
| "learning_rate": 0.0006392, |
| "loss": 4.8805, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 0.497113436460495, |
| "learning_rate": 0.0007191999999999999, |
| "loss": 4.7207, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.47698545455932617, |
| "learning_rate": 0.0007992, |
| "loss": 4.6068, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_accuracy": 0.2703913894324853, |
| "eval_loss": 4.5053935050964355, |
| "eval_runtime": 13.0992, |
| "eval_samples_per_second": 76.341, |
| "eval_steps_per_second": 0.611, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 0.44341349601745605, |
| "learning_rate": 0.0008792, |
| "loss": 4.4904, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.4383724331855774, |
| "learning_rate": 0.0009592000000000001, |
| "loss": 4.439, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 0.3894674777984619, |
| "learning_rate": 0.0010391999999999999, |
| "loss": 4.3735, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.41903820633888245, |
| "learning_rate": 0.0011192, |
| "loss": 4.3406, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 0.36601418256759644, |
| "learning_rate": 0.0011992, |
| "loss": 4.2879, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.4282030165195465, |
| "learning_rate": 0.0012791999999999999, |
| "loss": 4.2625, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 0.3431214690208435, |
| "learning_rate": 0.0013592, |
| "loss": 4.2316, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.3303460478782654, |
| "learning_rate": 0.0014392, |
| "loss": 4.1976, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 0.3030431866645813, |
| "learning_rate": 0.0015192, |
| "loss": 4.1494, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.3076256513595581, |
| "learning_rate": 0.0015992, |
| "loss": 4.1688, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.30658512720156555, |
| "eval_loss": 4.094621658325195, |
| "eval_runtime": 12.4723, |
| "eval_samples_per_second": 80.178, |
| "eval_steps_per_second": 0.641, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 0.2820931077003479, |
| "learning_rate": 0.0016792, |
| "loss": 4.5201, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.23724448680877686, |
| "learning_rate": 0.0017592, |
| "loss": 4.1333, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 0.2325439602136612, |
| "learning_rate": 0.0018392, |
| "loss": 4.1072, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.23736131191253662, |
| "learning_rate": 0.0019192, |
| "loss": 4.0682, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.2313733696937561, |
| "learning_rate": 0.0019992, |
| "loss": 4.0571, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.22318924963474274, |
| "learning_rate": 0.001999978563623903, |
| "loss": 4.0153, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 0.20275355875492096, |
| "learning_rate": 0.0019999133871331223, |
| "loss": 4.0163, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.21306632459163666, |
| "learning_rate": 0.0019998044711915177, |
| "loss": 3.9777, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 0.19690750539302826, |
| "learning_rate": 0.0019996518205634257, |
| "loss": 3.9603, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.20464760065078735, |
| "learning_rate": 0.0019994554419262797, |
| "loss": 3.9566, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.3226555772994129, |
| "eval_loss": 3.9112932682037354, |
| "eval_runtime": 11.7171, |
| "eval_samples_per_second": 85.345, |
| "eval_steps_per_second": 0.683, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 0.18151508271694183, |
| "learning_rate": 0.001999215343870317, |
| "loss": 3.9501, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 0.18748600780963898, |
| "learning_rate": 0.0019989315368982054, |
| "loss": 3.9228, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 0.17895972728729248, |
| "learning_rate": 0.0019986040334245797, |
| "loss": 3.9041, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 0.18976759910583496, |
| "learning_rate": 0.001998232847775504, |
| "loss": 3.9123, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 0.1871727555990219, |
| "learning_rate": 0.0019978179961878404, |
| "loss": 3.8846, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 0.17986708879470825, |
| "learning_rate": 0.001997359496808541, |
| "loss": 3.8794, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 0.17090782523155212, |
| "learning_rate": 0.001996857369693855, |
| "loss": 3.8593, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 0.19015273451805115, |
| "learning_rate": 0.0019963116368084486, |
| "loss": 3.8642, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 0.1871633529663086, |
| "learning_rate": 0.001995722322024446, |
| "loss": 3.8479, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 0.18049727380275726, |
| "learning_rate": 0.001995089451120385, |
| "loss": 3.8226, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.33194520547945205, |
| "eval_loss": 3.8037452697753906, |
| "eval_runtime": 12.6853, |
| "eval_samples_per_second": 78.832, |
| "eval_steps_per_second": 0.631, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.082, |
| "grad_norm": 0.14800746738910675, |
| "learning_rate": 0.0019944130517800893, |
| "loss": 3.8319, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.084, |
| "grad_norm": 0.14713813364505768, |
| "learning_rate": 0.001993693153591457, |
| "loss": 3.8319, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.086, |
| "grad_norm": 0.17019635438919067, |
| "learning_rate": 0.0019929297880451674, |
| "loss": 3.8095, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.088, |
| "grad_norm": 0.15243591368198395, |
| "learning_rate": 0.0019921229885333023, |
| "loss": 3.8003, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.09, |
| "grad_norm": 0.16134986281394958, |
| "learning_rate": 0.001991272790347886, |
| "loss": 3.814, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.092, |
| "grad_norm": 0.16883811354637146, |
| "learning_rate": 0.0019903792306793415, |
| "loss": 3.7822, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.094, |
| "grad_norm": 0.15718473494052887, |
| "learning_rate": 0.001989442348614863, |
| "loss": 3.7902, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.096, |
| "grad_norm": 0.14631035923957825, |
| "learning_rate": 0.0019884621851367075, |
| "loss": 3.7805, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.098, |
| "grad_norm": 0.14771369099617004, |
| "learning_rate": 0.001987438783120401, |
| "loss": 3.7778, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 0.1536298543214798, |
| "learning_rate": 0.001986372187332862, |
| "loss": 3.7848, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.1, |
| "eval_accuracy": 0.340252446183953, |
| "eval_loss": 3.7290549278259277, |
| "eval_runtime": 11.9375, |
| "eval_samples_per_second": 83.77, |
| "eval_steps_per_second": 0.67, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.102, |
| "grad_norm": 0.15110976994037628, |
| "learning_rate": 0.0019852624444304467, |
| "loss": 3.7502, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.104, |
| "grad_norm": 0.15862922370433807, |
| "learning_rate": 0.0019841096029569044, |
| "loss": 3.761, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.106, |
| "grad_norm": 0.13948658108711243, |
| "learning_rate": 0.0019829137133412556, |
| "loss": 3.7591, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.108, |
| "grad_norm": 0.148148313164711, |
| "learning_rate": 0.001981674827895587, |
| "loss": 3.7385, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.11, |
| "grad_norm": 0.13170774281024933, |
| "learning_rate": 0.00198039300081276, |
| "loss": 3.7424, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.112, |
| "grad_norm": 0.14352479577064514, |
| "learning_rate": 0.0019790682881640448, |
| "loss": 3.7423, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.114, |
| "grad_norm": 0.12865835428237915, |
| "learning_rate": 0.001977700747896664, |
| "loss": 3.7263, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.116, |
| "grad_norm": 0.15470698475837708, |
| "learning_rate": 0.001976290439831259, |
| "loss": 3.7297, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.118, |
| "grad_norm": 0.1353532075881958, |
| "learning_rate": 0.0019748374256592736, |
| "loss": 3.7218, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 0.14982788264751434, |
| "learning_rate": 0.0019733417689402543, |
| "loss": 3.7164, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.12, |
| "eval_accuracy": 0.34496673189823873, |
| "eval_loss": 3.6746184825897217, |
| "eval_runtime": 11.9904, |
| "eval_samples_per_second": 83.4, |
| "eval_steps_per_second": 0.667, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.122, |
| "grad_norm": 0.14931881427764893, |
| "learning_rate": 0.0019718035350990712, |
| "loss": 3.7169, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.124, |
| "grad_norm": 0.14251381158828735, |
| "learning_rate": 0.0019702227914230566, |
| "loss": 3.6929, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.126, |
| "grad_norm": 0.14251849055290222, |
| "learning_rate": 0.001968599607059059, |
| "loss": 3.7018, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 0.15696577727794647, |
| "learning_rate": 0.0019669340530104207, |
| "loss": 3.7034, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.13, |
| "grad_norm": 0.14737001061439514, |
| "learning_rate": 0.001965226202133872, |
| "loss": 3.6889, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.132, |
| "grad_norm": 0.13667502999305725, |
| "learning_rate": 0.0019634761291363427, |
| "loss": 3.6935, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.134, |
| "grad_norm": 0.13390734791755676, |
| "learning_rate": 0.0019616839105716954, |
| "loss": 3.7032, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.136, |
| "grad_norm": 0.14834214746952057, |
| "learning_rate": 0.0019598496248373755, |
| "loss": 3.6657, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.138, |
| "grad_norm": 0.1314767599105835, |
| "learning_rate": 0.001957973352170984, |
| "loss": 3.673, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 0.15161661803722382, |
| "learning_rate": 0.001956055174646765, |
| "loss": 3.6713, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.14, |
| "eval_accuracy": 0.3484422700587084, |
| "eval_loss": 3.6349546909332275, |
| "eval_runtime": 12.1265, |
| "eval_samples_per_second": 82.464, |
| "eval_steps_per_second": 0.66, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.142, |
| "grad_norm": 0.12710200250148773, |
| "learning_rate": 0.0019540951761720174, |
| "loss": 3.67, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.144, |
| "grad_norm": 0.18498213589191437, |
| "learning_rate": 0.0019520934424834247, |
| "loss": 3.675, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.146, |
| "grad_norm": 0.13697299361228943, |
| "learning_rate": 0.0019500500611433025, |
| "loss": 3.6497, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.148, |
| "grad_norm": 0.13290269672870636, |
| "learning_rate": 0.0019479651215357707, |
| "loss": 3.662, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 0.12744054198265076, |
| "learning_rate": 0.0019458387148628417, |
| "loss": 3.6645, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.152, |
| "grad_norm": 0.13697493076324463, |
| "learning_rate": 0.001943670934140432, |
| "loss": 3.6374, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.154, |
| "grad_norm": 0.14067181944847107, |
| "learning_rate": 0.0019414618741942936, |
| "loss": 3.6453, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.156, |
| "grad_norm": 0.1316055953502655, |
| "learning_rate": 0.0019392116316558638, |
| "loss": 3.6679, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.158, |
| "grad_norm": 0.1261526495218277, |
| "learning_rate": 0.001936920304958042, |
| "loss": 3.6343, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 0.13644501566886902, |
| "learning_rate": 0.0019345879943308804, |
| "loss": 3.6365, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.16, |
| "eval_accuracy": 0.35229158512720155, |
| "eval_loss": 3.5980424880981445, |
| "eval_runtime": 17.6098, |
| "eval_samples_per_second": 56.787, |
| "eval_steps_per_second": 0.454, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.162, |
| "grad_norm": 0.13911907374858856, |
| "learning_rate": 0.0019322148017972016, |
| "loss": 3.646, |
| "step": 8100 |
| }, |
| { |
| "epoch": 0.164, |
| "grad_norm": 0.1297680288553238, |
| "learning_rate": 0.001929800831168135, |
| "loss": 3.6412, |
| "step": 8200 |
| }, |
| { |
| "epoch": 0.166, |
| "grad_norm": 0.14803367853164673, |
| "learning_rate": 0.001927346188038576, |
| "loss": 3.6298, |
| "step": 8300 |
| }, |
| { |
| "epoch": 0.168, |
| "grad_norm": 0.1325535923242569, |
| "learning_rate": 0.0019248509797825672, |
| "loss": 3.6067, |
| "step": 8400 |
| }, |
| { |
| "epoch": 0.17, |
| "grad_norm": 0.16948860883712769, |
| "learning_rate": 0.0019223153155486009, |
| "loss": 3.6284, |
| "step": 8500 |
| }, |
| { |
| "epoch": 0.172, |
| "grad_norm": 0.12973898649215698, |
| "learning_rate": 0.0019197393062548454, |
| "loss": 3.627, |
| "step": 8600 |
| }, |
| { |
| "epoch": 0.174, |
| "grad_norm": 0.1576964110136032, |
| "learning_rate": 0.0019171230645842923, |
| "loss": 3.6027, |
| "step": 8700 |
| }, |
| { |
| "epoch": 0.176, |
| "grad_norm": 0.15394999086856842, |
| "learning_rate": 0.0019144667049798272, |
| "loss": 3.6098, |
| "step": 8800 |
| }, |
| { |
| "epoch": 0.178, |
| "grad_norm": 0.12594962120056152, |
| "learning_rate": 0.0019117703436392253, |
| "loss": 3.6221, |
| "step": 8900 |
| }, |
| { |
| "epoch": 0.18, |
| "grad_norm": 0.1420104056596756, |
| "learning_rate": 0.001909034098510066, |
| "loss": 3.5993, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.18, |
| "eval_accuracy": 0.3549628180039139, |
| "eval_loss": 3.570964813232422, |
| "eval_runtime": 12.4303, |
| "eval_samples_per_second": 80.449, |
| "eval_steps_per_second": 0.644, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.182, |
| "grad_norm": 0.12720872461795807, |
| "learning_rate": 0.001906258089284576, |
| "loss": 3.5913, |
| "step": 9100 |
| }, |
| { |
| "epoch": 0.184, |
| "grad_norm": 0.14890190958976746, |
| "learning_rate": 0.0019034424373943915, |
| "loss": 3.6113, |
| "step": 9200 |
| }, |
| { |
| "epoch": 0.186, |
| "grad_norm": 0.15023267269134521, |
| "learning_rate": 0.0019005872660052478, |
| "loss": 3.6011, |
| "step": 9300 |
| }, |
| { |
| "epoch": 0.188, |
| "grad_norm": 0.1637164056301117, |
| "learning_rate": 0.001897692700011591, |
| "loss": 3.5932, |
| "step": 9400 |
| }, |
| { |
| "epoch": 0.19, |
| "grad_norm": 0.13102982938289642, |
| "learning_rate": 0.0018947588660311143, |
| "loss": 3.5889, |
| "step": 9500 |
| }, |
| { |
| "epoch": 0.192, |
| "grad_norm": 0.16267850995063782, |
| "learning_rate": 0.0018917858923992211, |
| "loss": 3.5939, |
| "step": 9600 |
| }, |
| { |
| "epoch": 0.194, |
| "grad_norm": 0.12798982858657837, |
| "learning_rate": 0.0018887739091634085, |
| "loss": 3.5896, |
| "step": 9700 |
| }, |
| { |
| "epoch": 0.196, |
| "grad_norm": 0.13872383534908295, |
| "learning_rate": 0.0018857230480775807, |
| "loss": 3.574, |
| "step": 9800 |
| }, |
| { |
| "epoch": 0.198, |
| "grad_norm": 0.11993825435638428, |
| "learning_rate": 0.0018826334425962855, |
| "loss": 3.589, |
| "step": 9900 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 0.13074541091918945, |
| "learning_rate": 0.0018795052278688753, |
| "loss": 3.5828, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.2, |
| "eval_accuracy": 0.35816046966731896, |
| "eval_loss": 3.540745973587036, |
| "eval_runtime": 17.8387, |
| "eval_samples_per_second": 56.058, |
| "eval_steps_per_second": 0.448, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.202, |
| "grad_norm": 0.13834109902381897, |
| "learning_rate": 0.0018763385407335963, |
| "loss": 3.5627, |
| "step": 10100 |
| }, |
| { |
| "epoch": 0.204, |
| "grad_norm": 0.15819378197193146, |
| "learning_rate": 0.001873133519711602, |
| "loss": 3.5634, |
| "step": 10200 |
| }, |
| { |
| "epoch": 0.206, |
| "grad_norm": 0.1618553251028061, |
| "learning_rate": 0.0018698903050008956, |
| "loss": 3.5768, |
| "step": 10300 |
| }, |
| { |
| "epoch": 0.208, |
| "grad_norm": 0.14207501709461212, |
| "learning_rate": 0.0018666090384701947, |
| "loss": 3.5676, |
| "step": 10400 |
| }, |
| { |
| "epoch": 0.21, |
| "grad_norm": 0.13513191044330597, |
| "learning_rate": 0.001863289863652727, |
| "loss": 3.5653, |
| "step": 10500 |
| }, |
| { |
| "epoch": 0.212, |
| "grad_norm": 0.12555710971355438, |
| "learning_rate": 0.0018599329257399516, |
| "loss": 3.5577, |
| "step": 10600 |
| }, |
| { |
| "epoch": 0.214, |
| "grad_norm": 0.129461869597435, |
| "learning_rate": 0.0018565383715752083, |
| "loss": 3.576, |
| "step": 10700 |
| }, |
| { |
| "epoch": 0.216, |
| "grad_norm": 0.11782678216695786, |
| "learning_rate": 0.0018531063496472927, |
| "loss": 3.5695, |
| "step": 10800 |
| }, |
| { |
| "epoch": 0.218, |
| "grad_norm": 0.1391923427581787, |
| "learning_rate": 0.0018496370100839622, |
| "loss": 3.5478, |
| "step": 10900 |
| }, |
| { |
| "epoch": 0.22, |
| "grad_norm": 0.15454861521720886, |
| "learning_rate": 0.0018461305046453683, |
| "loss": 3.56, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.22, |
| "eval_accuracy": 0.3610430528375734, |
| "eval_loss": 3.5181522369384766, |
| "eval_runtime": 12.5872, |
| "eval_samples_per_second": 79.446, |
| "eval_steps_per_second": 0.636, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.222, |
| "grad_norm": 0.14800433814525604, |
| "learning_rate": 0.0018425869867174187, |
| "loss": 3.5559, |
| "step": 11100 |
| }, |
| { |
| "epoch": 0.224, |
| "grad_norm": 0.16677607595920563, |
| "learning_rate": 0.0018390066113050665, |
| "loss": 3.5393, |
| "step": 11200 |
| }, |
| { |
| "epoch": 0.226, |
| "grad_norm": 0.17119687795639038, |
| "learning_rate": 0.0018353895350255317, |
| "loss": 3.5503, |
| "step": 11300 |
| }, |
| { |
| "epoch": 0.228, |
| "grad_norm": 0.16998586058616638, |
| "learning_rate": 0.0018317359161014477, |
| "loss": 3.5546, |
| "step": 11400 |
| }, |
| { |
| "epoch": 0.23, |
| "grad_norm": 0.16900590062141418, |
| "learning_rate": 0.001828045914353943, |
| "loss": 3.5485, |
| "step": 11500 |
| }, |
| { |
| "epoch": 0.232, |
| "grad_norm": 0.13857169449329376, |
| "learning_rate": 0.0018243196911956476, |
| "loss": 3.5357, |
| "step": 11600 |
| }, |
| { |
| "epoch": 0.234, |
| "grad_norm": 0.134566068649292, |
| "learning_rate": 0.0018205574096236336, |
| "loss": 3.5478, |
| "step": 11700 |
| }, |
| { |
| "epoch": 0.236, |
| "grad_norm": 0.13580797612667084, |
| "learning_rate": 0.0018167592342122857, |
| "loss": 3.5452, |
| "step": 11800 |
| }, |
| { |
| "epoch": 0.238, |
| "grad_norm": 0.16239416599273682, |
| "learning_rate": 0.0018129253311061002, |
| "loss": 3.5256, |
| "step": 11900 |
| }, |
| { |
| "epoch": 0.24, |
| "grad_norm": 0.16310077905654907, |
| "learning_rate": 0.0018090558680124193, |
| "loss": 3.5331, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.24, |
| "eval_accuracy": 0.3630156555772994, |
| "eval_loss": 3.4982736110687256, |
| "eval_runtime": 12.3123, |
| "eval_samples_per_second": 81.22, |
| "eval_steps_per_second": 0.65, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.242, |
| "grad_norm": 0.1448379009962082, |
| "learning_rate": 0.0018051510141940939, |
| "loss": 3.5416, |
| "step": 12100 |
| }, |
| { |
| "epoch": 0.244, |
| "grad_norm": 0.13583365082740784, |
| "learning_rate": 0.00180121094046208, |
| "loss": 3.5343, |
| "step": 12200 |
| }, |
| { |
| "epoch": 0.246, |
| "grad_norm": 0.17297838628292084, |
| "learning_rate": 0.001797235819167967, |
| "loss": 3.521, |
| "step": 12300 |
| }, |
| { |
| "epoch": 0.248, |
| "grad_norm": 0.15616372227668762, |
| "learning_rate": 0.0017932258241964375, |
| "loss": 3.533, |
| "step": 12400 |
| }, |
| { |
| "epoch": 0.25, |
| "grad_norm": 0.15943922102451324, |
| "learning_rate": 0.0017891811309576622, |
| "loss": 3.5177, |
| "step": 12500 |
| }, |
| { |
| "epoch": 0.252, |
| "grad_norm": 0.15389597415924072, |
| "learning_rate": 0.001785101916379627, |
| "loss": 3.5209, |
| "step": 12600 |
| }, |
| { |
| "epoch": 0.254, |
| "grad_norm": 0.13968117535114288, |
| "learning_rate": 0.0017809883589003912, |
| "loss": 3.5137, |
| "step": 12700 |
| }, |
| { |
| "epoch": 0.256, |
| "grad_norm": 0.1350937783718109, |
| "learning_rate": 0.0017768406384602862, |
| "loss": 3.5294, |
| "step": 12800 |
| }, |
| { |
| "epoch": 0.258, |
| "grad_norm": 0.13691306114196777, |
| "learning_rate": 0.00177265893649404, |
| "loss": 3.5228, |
| "step": 12900 |
| }, |
| { |
| "epoch": 0.26, |
| "grad_norm": 0.1496269851922989, |
| "learning_rate": 0.0017684434359228438, |
| "loss": 3.5026, |
| "step": 13000 |
| }, |
| { |
| "epoch": 0.26, |
| "eval_accuracy": 0.3645440313111546, |
| "eval_loss": 3.480961799621582, |
| "eval_runtime": 11.914, |
| "eval_samples_per_second": 83.935, |
| "eval_steps_per_second": 0.671, |
| "step": 13000 |
| }, |
| { |
| "epoch": 0.262, |
| "grad_norm": 0.1472535878419876, |
| "learning_rate": 0.0017641943211463488, |
| "loss": 3.5107, |
| "step": 13100 |
| }, |
| { |
| "epoch": 0.264, |
| "grad_norm": 0.1573815494775772, |
| "learning_rate": 0.0017599117780346006, |
| "loss": 3.5147, |
| "step": 13200 |
| }, |
| { |
| "epoch": 0.266, |
| "grad_norm": 0.15841859579086304, |
| "learning_rate": 0.0017555959939199084, |
| "loss": 3.501, |
| "step": 13300 |
| }, |
| { |
| "epoch": 0.268, |
| "grad_norm": 0.15096184611320496, |
| "learning_rate": 0.0017512471575886505, |
| "loss": 3.5034, |
| "step": 13400 |
| }, |
| { |
| "epoch": 0.27, |
| "grad_norm": 0.17669688165187836, |
| "learning_rate": 0.0017468654592730168, |
| "loss": 3.5101, |
| "step": 13500 |
| }, |
| { |
| "epoch": 0.272, |
| "grad_norm": 0.16740451753139496, |
| "learning_rate": 0.0017424510906426853, |
| "loss": 3.5021, |
| "step": 13600 |
| }, |
| { |
| "epoch": 0.274, |
| "grad_norm": 0.1432366967201233, |
| "learning_rate": 0.0017380042447964414, |
| "loss": 3.4915, |
| "step": 13700 |
| }, |
| { |
| "epoch": 0.276, |
| "grad_norm": 0.14669281244277954, |
| "learning_rate": 0.0017335251162537279, |
| "loss": 3.4975, |
| "step": 13800 |
| }, |
| { |
| "epoch": 0.278, |
| "grad_norm": 0.15456849336624146, |
| "learning_rate": 0.0017290139009461377, |
| "loss": 3.504, |
| "step": 13900 |
| }, |
| { |
| "epoch": 0.28, |
| "grad_norm": 0.163591668009758, |
| "learning_rate": 0.0017244707962088424, |
| "loss": 3.4899, |
| "step": 14000 |
| }, |
| { |
| "epoch": 0.28, |
| "eval_accuracy": 0.366587084148728, |
| "eval_loss": 3.4621543884277344, |
| "eval_runtime": 13.6803, |
| "eval_samples_per_second": 73.098, |
| "eval_steps_per_second": 0.585, |
| "step": 14000 |
| }, |
| { |
| "epoch": 0.282, |
| "grad_norm": 0.1512901335954666, |
| "learning_rate": 0.0017198960007719611, |
| "loss": 3.4777, |
| "step": 14100 |
| }, |
| { |
| "epoch": 0.284, |
| "grad_norm": 0.13871945440769196, |
| "learning_rate": 0.0017152897147518665, |
| "loss": 3.5067, |
| "step": 14200 |
| }, |
| { |
| "epoch": 0.286, |
| "grad_norm": 0.15246713161468506, |
| "learning_rate": 0.001710652139642431, |
| "loss": 3.5018, |
| "step": 14300 |
| }, |
| { |
| "epoch": 0.288, |
| "grad_norm": 0.13609080016613007, |
| "learning_rate": 0.0017059834783062142, |
| "loss": 3.4762, |
| "step": 14400 |
| }, |
| { |
| "epoch": 0.29, |
| "grad_norm": 0.15287351608276367, |
| "learning_rate": 0.0017012839349655868, |
| "loss": 3.4872, |
| "step": 14500 |
| }, |
| { |
| "epoch": 0.292, |
| "grad_norm": 0.14040616154670715, |
| "learning_rate": 0.001696553715193799, |
| "loss": 3.5042, |
| "step": 14600 |
| }, |
| { |
| "epoch": 0.294, |
| "grad_norm": 0.1571398377418518, |
| "learning_rate": 0.0016917930259059879, |
| "loss": 3.4744, |
| "step": 14700 |
| }, |
| { |
| "epoch": 0.296, |
| "grad_norm": 0.1505391150712967, |
| "learning_rate": 0.001687002075350125, |
| "loss": 3.4749, |
| "step": 14800 |
| }, |
| { |
| "epoch": 0.298, |
| "grad_norm": 0.14757487177848816, |
| "learning_rate": 0.001682181073097908, |
| "loss": 3.4852, |
| "step": 14900 |
| }, |
| { |
| "epoch": 0.3, |
| "grad_norm": 0.15271714329719543, |
| "learning_rate": 0.0016773302300355935, |
| "loss": 3.486, |
| "step": 15000 |
| }, |
| { |
| "epoch": 0.3, |
| "eval_accuracy": 0.36803326810176124, |
| "eval_loss": 3.4481563568115234, |
| "eval_runtime": 12.0908, |
| "eval_samples_per_second": 82.707, |
| "eval_steps_per_second": 0.662, |
| "step": 15000 |
| }, |
| { |
| "epoch": 0.302, |
| "grad_norm": 0.15327778458595276, |
| "learning_rate": 0.0016724497583547715, |
| "loss": 3.478, |
| "step": 15100 |
| }, |
| { |
| "epoch": 0.304, |
| "grad_norm": 0.16108167171478271, |
| "learning_rate": 0.0016675398715430842, |
| "loss": 3.4579, |
| "step": 15200 |
| }, |
| { |
| "epoch": 0.306, |
| "grad_norm": 0.15231835842132568, |
| "learning_rate": 0.001662600784374886, |
| "loss": 3.4819, |
| "step": 15300 |
| }, |
| { |
| "epoch": 0.308, |
| "grad_norm": 0.1505262553691864, |
| "learning_rate": 0.0016576327129018504, |
| "loss": 3.4808, |
| "step": 15400 |
| }, |
| { |
| "epoch": 0.31, |
| "grad_norm": 0.15296302735805511, |
| "learning_rate": 0.0016526358744435184, |
| "loss": 3.459, |
| "step": 15500 |
| }, |
| { |
| "epoch": 0.312, |
| "grad_norm": 0.2003161907196045, |
| "learning_rate": 0.001647610487577791, |
| "loss": 3.4751, |
| "step": 15600 |
| }, |
| { |
| "epoch": 0.314, |
| "grad_norm": 0.130265474319458, |
| "learning_rate": 0.0016425567721313694, |
| "loss": 3.4819, |
| "step": 15700 |
| }, |
| { |
| "epoch": 0.316, |
| "grad_norm": 0.14476525783538818, |
| "learning_rate": 0.0016374749491701401, |
| "loss": 3.4514, |
| "step": 15800 |
| }, |
| { |
| "epoch": 0.318, |
| "grad_norm": 0.1549970805644989, |
| "learning_rate": 0.0016323652409895013, |
| "loss": 3.4684, |
| "step": 15900 |
| }, |
| { |
| "epoch": 0.32, |
| "grad_norm": 0.14814923703670502, |
| "learning_rate": 0.0016272278711046424, |
| "loss": 3.4674, |
| "step": 16000 |
| }, |
| { |
| "epoch": 0.32, |
| "eval_accuracy": 0.36914090019569473, |
| "eval_loss": 3.435351848602295, |
| "eval_runtime": 12.375, |
| "eval_samples_per_second": 80.808, |
| "eval_steps_per_second": 0.646, |
| "step": 16000 |
| }, |
| { |
| "epoch": 0.322, |
| "grad_norm": 0.207662433385849, |
| "learning_rate": 0.001622063064240765, |
| "loss": 3.4661, |
| "step": 16100 |
| }, |
| { |
| "epoch": 0.324, |
| "grad_norm": 0.14697088301181793, |
| "learning_rate": 0.001616871046323253, |
| "loss": 3.4735, |
| "step": 16200 |
| }, |
| { |
| "epoch": 0.326, |
| "grad_norm": 0.16137182712554932, |
| "learning_rate": 0.0016116520444677894, |
| "loss": 3.451, |
| "step": 16300 |
| }, |
| { |
| "epoch": 0.328, |
| "grad_norm": 0.1839095950126648, |
| "learning_rate": 0.001606406286970423, |
| "loss": 3.4646, |
| "step": 16400 |
| }, |
| { |
| "epoch": 0.33, |
| "grad_norm": 0.1534615308046341, |
| "learning_rate": 0.0016011340032975805, |
| "loss": 3.4666, |
| "step": 16500 |
| }, |
| { |
| "epoch": 0.332, |
| "grad_norm": 0.15798258781433105, |
| "learning_rate": 0.00159583542407603, |
| "loss": 3.4418, |
| "step": 16600 |
| }, |
| { |
| "epoch": 0.334, |
| "grad_norm": 0.13258863985538483, |
| "learning_rate": 0.001590510781082791, |
| "loss": 3.4641, |
| "step": 16700 |
| }, |
| { |
| "epoch": 0.336, |
| "grad_norm": 0.14669552445411682, |
| "learning_rate": 0.001585160307234998, |
| "loss": 3.4667, |
| "step": 16800 |
| }, |
| { |
| "epoch": 0.338, |
| "grad_norm": 0.15622150897979736, |
| "learning_rate": 0.00157978423657971, |
| "loss": 3.439, |
| "step": 16900 |
| }, |
| { |
| "epoch": 0.34, |
| "grad_norm": 0.18012835085391998, |
| "learning_rate": 0.0015743828042836733, |
| "loss": 3.4512, |
| "step": 17000 |
| }, |
| { |
| "epoch": 0.34, |
| "eval_accuracy": 0.3706555772994129, |
| "eval_loss": 3.4220447540283203, |
| "eval_runtime": 21.6548, |
| "eval_samples_per_second": 46.179, |
| "eval_steps_per_second": 0.369, |
| "step": 17000 |
| }, |
| { |
| "epoch": 0.342, |
| "grad_norm": 0.143183171749115, |
| "learning_rate": 0.0015689562466230352, |
| "loss": 3.4531, |
| "step": 17100 |
| }, |
| { |
| "epoch": 0.344, |
| "grad_norm": 0.15830625593662262, |
| "learning_rate": 0.0015635048009730072, |
| "loss": 3.4425, |
| "step": 17200 |
| }, |
| { |
| "epoch": 0.346, |
| "grad_norm": 0.1351785659790039, |
| "learning_rate": 0.0015580287057974822, |
| "loss": 3.4491, |
| "step": 17300 |
| }, |
| { |
| "epoch": 0.348, |
| "grad_norm": 0.1572723239660263, |
| "learning_rate": 0.0015525282006386032, |
| "loss": 3.4329, |
| "step": 17400 |
| }, |
| { |
| "epoch": 0.35, |
| "grad_norm": 0.16905032098293304, |
| "learning_rate": 0.0015470035261062852, |
| "loss": 3.4457, |
| "step": 17500 |
| }, |
| { |
| "epoch": 0.352, |
| "grad_norm": 0.1542733907699585, |
| "learning_rate": 0.0015414549238676899, |
| "loss": 3.4466, |
| "step": 17600 |
| }, |
| { |
| "epoch": 0.354, |
| "grad_norm": 0.17319922149181366, |
| "learning_rate": 0.0015358826366366532, |
| "loss": 3.4311, |
| "step": 17700 |
| }, |
| { |
| "epoch": 0.356, |
| "grad_norm": 0.17198143899440765, |
| "learning_rate": 0.0015302869081630717, |
| "loss": 3.4453, |
| "step": 17800 |
| }, |
| { |
| "epoch": 0.358, |
| "grad_norm": 0.15461337566375732, |
| "learning_rate": 0.0015246679832222353, |
| "loss": 3.4539, |
| "step": 17900 |
| }, |
| { |
| "epoch": 0.36, |
| "grad_norm": 0.15917198359966278, |
| "learning_rate": 0.0015190261076041245, |
| "loss": 3.4188, |
| "step": 18000 |
| }, |
| { |
| "epoch": 0.36, |
| "eval_accuracy": 0.3723365949119374, |
| "eval_loss": 3.4081637859344482, |
| "eval_runtime": 12.6994, |
| "eval_samples_per_second": 78.744, |
| "eval_steps_per_second": 0.63, |
| "step": 18000 |
| }, |
| { |
| "epoch": 0.362, |
| "grad_norm": 0.15682971477508545, |
| "learning_rate": 0.0015133615281026555, |
| "loss": 3.426, |
| "step": 18100 |
| }, |
| { |
| "epoch": 0.364, |
| "grad_norm": 0.15073339641094208, |
| "learning_rate": 0.0015076744925048868, |
| "loss": 3.4358, |
| "step": 18200 |
| }, |
| { |
| "epoch": 0.366, |
| "grad_norm": 0.18040859699249268, |
| "learning_rate": 0.0015019652495801786, |
| "loss": 3.4261, |
| "step": 18300 |
| }, |
| { |
| "epoch": 0.368, |
| "grad_norm": 0.18911828100681305, |
| "learning_rate": 0.0014962340490693121, |
| "loss": 3.4359, |
| "step": 18400 |
| }, |
| { |
| "epoch": 0.37, |
| "grad_norm": 0.15816016495227814, |
| "learning_rate": 0.0014904811416735636, |
| "loss": 3.4216, |
| "step": 18500 |
| }, |
| { |
| "epoch": 0.372, |
| "grad_norm": 0.17654721438884735, |
| "learning_rate": 0.0014847067790437396, |
| "loss": 3.4347, |
| "step": 18600 |
| }, |
| { |
| "epoch": 0.374, |
| "grad_norm": 0.17527590692043304, |
| "learning_rate": 0.0014789112137691682, |
| "loss": 3.4256, |
| "step": 18700 |
| }, |
| { |
| "epoch": 0.376, |
| "grad_norm": 0.15412954986095428, |
| "learning_rate": 0.001473094699366649, |
| "loss": 3.4074, |
| "step": 18800 |
| }, |
| { |
| "epoch": 0.378, |
| "grad_norm": 0.1891617327928543, |
| "learning_rate": 0.0014672574902693661, |
| "loss": 3.4338, |
| "step": 18900 |
| }, |
| { |
| "epoch": 0.38, |
| "grad_norm": 0.16060151159763336, |
| "learning_rate": 0.001461399841815754, |
| "loss": 3.4372, |
| "step": 19000 |
| }, |
| { |
| "epoch": 0.38, |
| "eval_accuracy": 0.3734951076320939, |
| "eval_loss": 3.3982865810394287, |
| "eval_runtime": 12.1716, |
| "eval_samples_per_second": 82.158, |
| "eval_steps_per_second": 0.657, |
| "step": 19000 |
| }, |
| { |
| "epoch": 0.382, |
| "grad_norm": 0.17226573824882507, |
| "learning_rate": 0.0014555220102383326, |
| "loss": 3.4123, |
| "step": 19100 |
| }, |
| { |
| "epoch": 0.384, |
| "grad_norm": 0.15569022297859192, |
| "learning_rate": 0.0014496242526524964, |
| "loss": 3.4226, |
| "step": 19200 |
| }, |
| { |
| "epoch": 0.386, |
| "grad_norm": 0.16819970309734344, |
| "learning_rate": 0.001443706827045269, |
| "loss": 3.4294, |
| "step": 19300 |
| }, |
| { |
| "epoch": 0.388, |
| "grad_norm": 0.14779409766197205, |
| "learning_rate": 0.0014377699922640157, |
| "loss": 3.4233, |
| "step": 19400 |
| }, |
| { |
| "epoch": 0.39, |
| "grad_norm": 0.1541208177804947, |
| "learning_rate": 0.0014318140080051228, |
| "loss": 3.4087, |
| "step": 19500 |
| }, |
| { |
| "epoch": 0.392, |
| "grad_norm": 0.1646755486726761, |
| "learning_rate": 0.0014258391348026362, |
| "loss": 3.398, |
| "step": 19600 |
| }, |
| { |
| "epoch": 0.394, |
| "grad_norm": 0.18935076892375946, |
| "learning_rate": 0.0014198456340168662, |
| "loss": 3.4231, |
| "step": 19700 |
| }, |
| { |
| "epoch": 0.396, |
| "grad_norm": 0.1867913156747818, |
| "learning_rate": 0.0014138337678229528, |
| "loss": 3.4129, |
| "step": 19800 |
| }, |
| { |
| "epoch": 0.398, |
| "grad_norm": 0.15837714076042175, |
| "learning_rate": 0.0014078037991993992, |
| "loss": 3.3952, |
| "step": 19900 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 0.15315821766853333, |
| "learning_rate": 0.0014017559919165675, |
| "loss": 3.4155, |
| "step": 20000 |
| }, |
| { |
| "epoch": 0.4, |
| "eval_accuracy": 0.37439138943248534, |
| "eval_loss": 3.387746572494507, |
| "eval_runtime": 12.7148, |
| "eval_samples_per_second": 78.648, |
| "eval_steps_per_second": 0.629, |
| "step": 20000 |
| }, |
| { |
| "epoch": 0.402, |
| "grad_norm": 0.18359500169754028, |
| "learning_rate": 0.0013956906105251406, |
| "loss": 3.4094, |
| "step": 20100 |
| }, |
| { |
| "epoch": 0.404, |
| "grad_norm": 0.1576140820980072, |
| "learning_rate": 0.0013896079203445497, |
| "loss": 3.3935, |
| "step": 20200 |
| }, |
| { |
| "epoch": 0.406, |
| "grad_norm": 0.22549428045749664, |
| "learning_rate": 0.0013835081874513679, |
| "loss": 3.4006, |
| "step": 20300 |
| }, |
| { |
| "epoch": 0.408, |
| "grad_norm": 0.15331390500068665, |
| "learning_rate": 0.001377391678667673, |
| "loss": 3.4129, |
| "step": 20400 |
| }, |
| { |
| "epoch": 0.41, |
| "grad_norm": 0.16119052469730377, |
| "learning_rate": 0.0013712586615493736, |
| "loss": 3.4036, |
| "step": 20500 |
| }, |
| { |
| "epoch": 0.412, |
| "grad_norm": 0.15083439648151398, |
| "learning_rate": 0.0013651094043745067, |
| "loss": 3.3857, |
| "step": 20600 |
| }, |
| { |
| "epoch": 0.414, |
| "grad_norm": 0.14811307191848755, |
| "learning_rate": 0.0013589441761315021, |
| "loss": 3.4058, |
| "step": 20700 |
| }, |
| { |
| "epoch": 0.416, |
| "grad_norm": 0.15205927193164825, |
| "learning_rate": 0.0013527632465074157, |
| "loss": 3.4109, |
| "step": 20800 |
| }, |
| { |
| "epoch": 0.418, |
| "grad_norm": 0.164772629737854, |
| "learning_rate": 0.0013465668858761326, |
| "loss": 3.3811, |
| "step": 20900 |
| }, |
| { |
| "epoch": 0.42, |
| "grad_norm": 0.15865328907966614, |
| "learning_rate": 0.00134035536528654, |
| "loss": 3.3878, |
| "step": 21000 |
| }, |
| { |
| "epoch": 0.42, |
| "eval_accuracy": 0.37497260273972605, |
| "eval_loss": 3.3792359828948975, |
| "eval_runtime": 12.1681, |
| "eval_samples_per_second": 82.182, |
| "eval_steps_per_second": 0.657, |
| "step": 21000 |
| }, |
| { |
| "epoch": 0.422, |
| "grad_norm": 0.18703462183475494, |
| "learning_rate": 0.0013341289564506717, |
| "loss": 3.401, |
| "step": 21100 |
| }, |
| { |
| "epoch": 0.424, |
| "grad_norm": 0.15354931354522705, |
| "learning_rate": 0.0013278879317318202, |
| "loss": 3.3947, |
| "step": 21200 |
| }, |
| { |
| "epoch": 0.426, |
| "grad_norm": 0.14713342487812042, |
| "learning_rate": 0.0013216325641326255, |
| "loss": 3.3774, |
| "step": 21300 |
| }, |
| { |
| "epoch": 0.428, |
| "grad_norm": 0.17548827826976776, |
| "learning_rate": 0.0013153631272831304, |
| "loss": 3.3794, |
| "step": 21400 |
| }, |
| { |
| "epoch": 0.43, |
| "grad_norm": 0.1897333562374115, |
| "learning_rate": 0.001309079895428813, |
| "loss": 3.3955, |
| "step": 21500 |
| }, |
| { |
| "epoch": 0.432, |
| "grad_norm": 0.1576170176267624, |
| "learning_rate": 0.0013027831434185898, |
| "loss": 3.3884, |
| "step": 21600 |
| }, |
| { |
| "epoch": 0.434, |
| "grad_norm": 0.18555521965026855, |
| "learning_rate": 0.0012964731466927916, |
| "loss": 3.3695, |
| "step": 21700 |
| }, |
| { |
| "epoch": 0.436, |
| "grad_norm": 0.15987864136695862, |
| "learning_rate": 0.0012901501812711176, |
| "loss": 3.3891, |
| "step": 21800 |
| }, |
| { |
| "epoch": 0.438, |
| "grad_norm": 0.16719584167003632, |
| "learning_rate": 0.0012838145237405588, |
| "loss": 3.4056, |
| "step": 21900 |
| }, |
| { |
| "epoch": 0.44, |
| "grad_norm": 0.19731645286083221, |
| "learning_rate": 0.0012774664512432996, |
| "loss": 3.3772, |
| "step": 22000 |
| }, |
| { |
| "epoch": 0.44, |
| "eval_accuracy": 0.3771624266144814, |
| "eval_loss": 3.3640236854553223, |
| "eval_runtime": 12.4106, |
| "eval_samples_per_second": 80.576, |
| "eval_steps_per_second": 0.645, |
| "step": 22000 |
| }, |
| { |
| "epoch": 0.442, |
| "grad_norm": 0.1659466177225113, |
| "learning_rate": 0.0012711062414645967, |
| "loss": 3.3785, |
| "step": 22100 |
| }, |
| { |
| "epoch": 0.444, |
| "grad_norm": 0.15995340049266815, |
| "learning_rate": 0.0012647341726206296, |
| "loss": 3.389, |
| "step": 22200 |
| }, |
| { |
| "epoch": 0.446, |
| "grad_norm": 0.17644192278385162, |
| "learning_rate": 0.0012583505234463322, |
| "loss": 3.3787, |
| "step": 22300 |
| }, |
| { |
| "epoch": 0.448, |
| "grad_norm": 0.170969158411026, |
| "learning_rate": 0.0012519555731831996, |
| "loss": 3.365, |
| "step": 22400 |
| }, |
| { |
| "epoch": 0.45, |
| "grad_norm": 0.1832742989063263, |
| "learning_rate": 0.0012455496015670732, |
| "loss": 3.3835, |
| "step": 22500 |
| }, |
| { |
| "epoch": 0.452, |
| "grad_norm": 0.16345185041427612, |
| "learning_rate": 0.001239132888815904, |
| "loss": 3.3824, |
| "step": 22600 |
| }, |
| { |
| "epoch": 0.454, |
| "grad_norm": 0.16044080257415771, |
| "learning_rate": 0.0012327057156174945, |
| "loss": 3.3691, |
| "step": 22700 |
| }, |
| { |
| "epoch": 0.456, |
| "grad_norm": 0.1500287652015686, |
| "learning_rate": 0.0012262683631172222, |
| "loss": 3.3775, |
| "step": 22800 |
| }, |
| { |
| "epoch": 0.458, |
| "grad_norm": 0.1463031768798828, |
| "learning_rate": 0.0012198211129057393, |
| "loss": 3.377, |
| "step": 22900 |
| }, |
| { |
| "epoch": 0.46, |
| "grad_norm": 0.15354429185390472, |
| "learning_rate": 0.0012133642470066562, |
| "loss": 3.3741, |
| "step": 23000 |
| }, |
| { |
| "epoch": 0.46, |
| "eval_accuracy": 0.3781017612524462, |
| "eval_loss": 3.3543314933776855, |
| "eval_runtime": 14.9495, |
| "eval_samples_per_second": 66.892, |
| "eval_steps_per_second": 0.535, |
| "step": 23000 |
| }, |
| { |
| "epoch": 0.462, |
| "grad_norm": 0.16891853511333466, |
| "learning_rate": 0.0012068980478642047, |
| "loss": 3.3573, |
| "step": 23100 |
| }, |
| { |
| "epoch": 0.464, |
| "grad_norm": 0.15053242444992065, |
| "learning_rate": 0.0012004227983308828, |
| "loss": 3.3767, |
| "step": 23200 |
| }, |
| { |
| "epoch": 0.466, |
| "grad_norm": 0.13770438730716705, |
| "learning_rate": 0.001193938781655082, |
| "loss": 3.3735, |
| "step": 23300 |
| }, |
| { |
| "epoch": 0.468, |
| "grad_norm": 0.16500090062618256, |
| "learning_rate": 0.0011874462814686975, |
| "loss": 3.3594, |
| "step": 23400 |
| }, |
| { |
| "epoch": 0.47, |
| "grad_norm": 0.15050701797008514, |
| "learning_rate": 0.0011809455817747194, |
| "loss": 3.3736, |
| "step": 23500 |
| }, |
| { |
| "epoch": 0.472, |
| "grad_norm": 0.14346493780612946, |
| "learning_rate": 0.0011744369669348122, |
| "loss": 3.3714, |
| "step": 23600 |
| }, |
| { |
| "epoch": 0.474, |
| "grad_norm": 0.15672604739665985, |
| "learning_rate": 0.0011679207216568738, |
| "loss": 3.3547, |
| "step": 23700 |
| }, |
| { |
| "epoch": 0.476, |
| "grad_norm": 0.20898281037807465, |
| "learning_rate": 0.0011613971309825828, |
| "loss": 3.3603, |
| "step": 23800 |
| }, |
| { |
| "epoch": 0.478, |
| "grad_norm": 0.15672914683818817, |
| "learning_rate": 0.001154866480274928, |
| "loss": 3.363, |
| "step": 23900 |
| }, |
| { |
| "epoch": 0.48, |
| "grad_norm": 0.15417250990867615, |
| "learning_rate": 0.0011483290552057285, |
| "loss": 3.3706, |
| "step": 24000 |
| }, |
| { |
| "epoch": 0.48, |
| "eval_accuracy": 0.37883365949119374, |
| "eval_loss": 3.344762086868286, |
| "eval_runtime": 12.8887, |
| "eval_samples_per_second": 77.587, |
| "eval_steps_per_second": 0.621, |
| "step": 24000 |
| }, |
| { |
| "epoch": 0.482, |
| "grad_norm": 0.17039372026920319, |
| "learning_rate": 0.0011417851417431346, |
| "loss": 3.3587, |
| "step": 24100 |
| }, |
| { |
| "epoch": 0.484, |
| "grad_norm": 0.16372039914131165, |
| "learning_rate": 0.0011352350261391204, |
| "loss": 3.3488, |
| "step": 24200 |
| }, |
| { |
| "epoch": 0.486, |
| "grad_norm": 0.15907979011535645, |
| "learning_rate": 0.0011286789949169628, |
| "loss": 3.3638, |
| "step": 24300 |
| }, |
| { |
| "epoch": 0.488, |
| "grad_norm": 0.15641772747039795, |
| "learning_rate": 0.001122117334858705, |
| "loss": 3.3627, |
| "step": 24400 |
| }, |
| { |
| "epoch": 0.49, |
| "grad_norm": 0.17920181155204773, |
| "learning_rate": 0.0011155503329926151, |
| "loss": 3.3278, |
| "step": 24500 |
| }, |
| { |
| "epoch": 0.492, |
| "grad_norm": 0.16552342474460602, |
| "learning_rate": 0.001108978276580629, |
| "loss": 3.3653, |
| "step": 24600 |
| }, |
| { |
| "epoch": 0.494, |
| "grad_norm": 0.17901502549648285, |
| "learning_rate": 0.001102401453105785, |
| "loss": 3.3626, |
| "step": 24700 |
| }, |
| { |
| "epoch": 0.496, |
| "grad_norm": 0.17022928595542908, |
| "learning_rate": 0.001095820150259647, |
| "loss": 3.3391, |
| "step": 24800 |
| }, |
| { |
| "epoch": 0.498, |
| "grad_norm": 0.1806282252073288, |
| "learning_rate": 0.0010892346559297226, |
| "loss": 3.3476, |
| "step": 24900 |
| }, |
| { |
| "epoch": 0.5, |
| "grad_norm": 0.16578249633312225, |
| "learning_rate": 0.0010826452581868676, |
| "loss": 3.3586, |
| "step": 25000 |
| }, |
| { |
| "epoch": 0.5, |
| "eval_accuracy": 0.38070841487279844, |
| "eval_loss": 3.3385705947875977, |
| "eval_runtime": 17.5524, |
| "eval_samples_per_second": 56.972, |
| "eval_steps_per_second": 0.456, |
| "step": 25000 |
| }, |
| { |
| "epoch": 0.502, |
| "grad_norm": 0.14317767322063446, |
| "learning_rate": 0.001076052245272686, |
| "loss": 3.3494, |
| "step": 25100 |
| }, |
| { |
| "epoch": 0.504, |
| "grad_norm": 0.18083012104034424, |
| "learning_rate": 0.0010694559055869205, |
| "loss": 3.3425, |
| "step": 25200 |
| }, |
| { |
| "epoch": 0.506, |
| "grad_norm": 0.1475485861301422, |
| "learning_rate": 0.0010628565276748394, |
| "loss": 3.3346, |
| "step": 25300 |
| }, |
| { |
| "epoch": 0.508, |
| "grad_norm": 0.19806760549545288, |
| "learning_rate": 0.0010562544002146108, |
| "loss": 3.359, |
| "step": 25400 |
| }, |
| { |
| "epoch": 0.51, |
| "grad_norm": 0.1636313647031784, |
| "learning_rate": 0.0010496498120046778, |
| "loss": 3.351, |
| "step": 25500 |
| }, |
| { |
| "epoch": 0.512, |
| "grad_norm": 0.14796622097492218, |
| "learning_rate": 0.0010430430519511244, |
| "loss": 3.3308, |
| "step": 25600 |
| }, |
| { |
| "epoch": 0.514, |
| "grad_norm": 0.15158209204673767, |
| "learning_rate": 0.001036434409055039, |
| "loss": 3.3462, |
| "step": 25700 |
| }, |
| { |
| "epoch": 0.516, |
| "grad_norm": 0.1509816199541092, |
| "learning_rate": 0.0010298241723998701, |
| "loss": 3.3612, |
| "step": 25800 |
| }, |
| { |
| "epoch": 0.518, |
| "grad_norm": 0.2039985954761505, |
| "learning_rate": 0.001023212631138783, |
| "loss": 3.3233, |
| "step": 25900 |
| }, |
| { |
| "epoch": 0.52, |
| "grad_norm": 0.15846829116344452, |
| "learning_rate": 0.001016600074482012, |
| "loss": 3.3338, |
| "step": 26000 |
| }, |
| { |
| "epoch": 0.52, |
| "eval_accuracy": 0.381559686888454, |
| "eval_loss": 3.330564498901367, |
| "eval_runtime": 11.5688, |
| "eval_samples_per_second": 86.439, |
| "eval_steps_per_second": 0.692, |
| "step": 26000 |
| }, |
| { |
| "epoch": 0.522, |
| "grad_norm": 0.16366560757160187, |
| "learning_rate": 0.0010099867916842063, |
| "loss": 3.349, |
| "step": 26100 |
| }, |
| { |
| "epoch": 0.524, |
| "grad_norm": 0.18745005130767822, |
| "learning_rate": 0.00100337307203178, |
| "loss": 3.3381, |
| "step": 26200 |
| }, |
| { |
| "epoch": 0.526, |
| "grad_norm": 0.14261376857757568, |
| "learning_rate": 0.0009967592048302552, |
| "loss": 3.334, |
| "step": 26300 |
| }, |
| { |
| "epoch": 0.528, |
| "grad_norm": 0.16403239965438843, |
| "learning_rate": 0.0009901454793916102, |
| "loss": 3.3253, |
| "step": 26400 |
| }, |
| { |
| "epoch": 0.53, |
| "grad_norm": 0.1447836309671402, |
| "learning_rate": 0.00098353218502162, |
| "loss": 3.3442, |
| "step": 26500 |
| }, |
| { |
| "epoch": 0.532, |
| "grad_norm": 0.18312039971351624, |
| "learning_rate": 0.0009769196110072061, |
| "loss": 3.338, |
| "step": 26600 |
| }, |
| { |
| "epoch": 0.534, |
| "grad_norm": 0.1727229803800583, |
| "learning_rate": 0.0009703080466037767, |
| "loss": 3.3183, |
| "step": 26700 |
| }, |
| { |
| "epoch": 0.536, |
| "grad_norm": 0.15190576016902924, |
| "learning_rate": 0.0009636977810225777, |
| "loss": 3.3387, |
| "step": 26800 |
| }, |
| { |
| "epoch": 0.538, |
| "grad_norm": 0.1397494077682495, |
| "learning_rate": 0.0009570891034180394, |
| "loss": 3.3415, |
| "step": 26900 |
| }, |
| { |
| "epoch": 0.54, |
| "grad_norm": 0.1609794944524765, |
| "learning_rate": 0.0009504823028751295, |
| "loss": 3.3174, |
| "step": 27000 |
| }, |
| { |
| "epoch": 0.54, |
| "eval_accuracy": 0.38231311154598824, |
| "eval_loss": 3.3193719387054443, |
| "eval_runtime": 11.9776, |
| "eval_samples_per_second": 83.489, |
| "eval_steps_per_second": 0.668, |
| "step": 27000 |
| }, |
| { |
| "epoch": 0.542, |
| "grad_norm": 0.14202991127967834, |
| "learning_rate": 0.0009438776683967071, |
| "loss": 3.334, |
| "step": 27100 |
| }, |
| { |
| "epoch": 0.544, |
| "grad_norm": 0.14222672581672668, |
| "learning_rate": 0.0009372754888908796, |
| "loss": 3.329, |
| "step": 27200 |
| }, |
| { |
| "epoch": 0.546, |
| "grad_norm": 0.2155715525150299, |
| "learning_rate": 0.000930676053158368, |
| "loss": 3.3328, |
| "step": 27300 |
| }, |
| { |
| "epoch": 0.548, |
| "grad_norm": 0.14550994336605072, |
| "learning_rate": 0.0009240796498798694, |
| "loss": 3.3314, |
| "step": 27400 |
| }, |
| { |
| "epoch": 0.55, |
| "grad_norm": 0.1446721851825714, |
| "learning_rate": 0.0009174865676034329, |
| "loss": 3.322, |
| "step": 27500 |
| }, |
| { |
| "epoch": 0.552, |
| "grad_norm": 0.15072402358055115, |
| "learning_rate": 0.000910897094731836, |
| "loss": 3.3292, |
| "step": 27600 |
| }, |
| { |
| "epoch": 0.554, |
| "grad_norm": 0.14199484884738922, |
| "learning_rate": 0.0009043115195099688, |
| "loss": 3.3319, |
| "step": 27700 |
| }, |
| { |
| "epoch": 0.556, |
| "grad_norm": 0.1559506356716156, |
| "learning_rate": 0.0008977301300122259, |
| "loss": 3.3161, |
| "step": 27800 |
| }, |
| { |
| "epoch": 0.558, |
| "grad_norm": 0.14755868911743164, |
| "learning_rate": 0.0008911532141299046, |
| "loss": 3.3283, |
| "step": 27900 |
| }, |
| { |
| "epoch": 0.56, |
| "grad_norm": 0.17486320436000824, |
| "learning_rate": 0.000884581059558612, |
| "loss": 3.3317, |
| "step": 28000 |
| }, |
| { |
| "epoch": 0.56, |
| "eval_accuracy": 0.38318786692759293, |
| "eval_loss": 3.3127212524414062, |
| "eval_runtime": 13.6461, |
| "eval_samples_per_second": 73.281, |
| "eval_steps_per_second": 0.586, |
| "step": 28000 |
| }, |
| { |
| "epoch": 0.562, |
| "grad_norm": 0.15159651637077332, |
| "learning_rate": 0.00087801395378568, |
| "loss": 3.3107, |
| "step": 28100 |
| }, |
| { |
| "epoch": 0.564, |
| "grad_norm": 0.16269639134407043, |
| "learning_rate": 0.0008714521840775893, |
| "loss": 3.3182, |
| "step": 28200 |
| }, |
| { |
| "epoch": 0.566, |
| "grad_norm": 0.16659656167030334, |
| "learning_rate": 0.0008648960374674042, |
| "loss": 3.3197, |
| "step": 28300 |
| }, |
| { |
| "epoch": 0.568, |
| "grad_norm": 0.1418466717004776, |
| "learning_rate": 0.0008583458007422165, |
| "loss": 3.3122, |
| "step": 28400 |
| }, |
| { |
| "epoch": 0.57, |
| "grad_norm": 0.14337651431560516, |
| "learning_rate": 0.0008518017604306002, |
| "loss": 3.3163, |
| "step": 28500 |
| }, |
| { |
| "epoch": 0.572, |
| "grad_norm": 0.16096735000610352, |
| "learning_rate": 0.0008452642027900783, |
| "loss": 3.3061, |
| "step": 28600 |
| }, |
| { |
| "epoch": 0.574, |
| "grad_norm": 0.18552157282829285, |
| "learning_rate": 0.0008387334137946008, |
| "loss": 3.318, |
| "step": 28700 |
| }, |
| { |
| "epoch": 0.576, |
| "grad_norm": 0.16036978363990784, |
| "learning_rate": 0.0008322096791220355, |
| "loss": 3.3109, |
| "step": 28800 |
| }, |
| { |
| "epoch": 0.578, |
| "grad_norm": 0.16824260354042053, |
| "learning_rate": 0.0008256932841416705, |
| "loss": 3.3074, |
| "step": 28900 |
| }, |
| { |
| "epoch": 0.58, |
| "grad_norm": 0.16867192089557648, |
| "learning_rate": 0.0008191845139017326, |
| "loss": 3.3225, |
| "step": 29000 |
| }, |
| { |
| "epoch": 0.58, |
| "eval_accuracy": 0.38399021526418786, |
| "eval_loss": 3.3025898933410645, |
| "eval_runtime": 11.9945, |
| "eval_samples_per_second": 83.372, |
| "eval_steps_per_second": 0.667, |
| "step": 29000 |
| }, |
| { |
| "epoch": 0.582, |
| "grad_norm": 0.1469780057668686, |
| "learning_rate": 0.0008126836531169177, |
| "loss": 3.3142, |
| "step": 29100 |
| }, |
| { |
| "epoch": 0.584, |
| "grad_norm": 0.15736377239227295, |
| "learning_rate": 0.0008061909861559362, |
| "loss": 3.2969, |
| "step": 29200 |
| }, |
| { |
| "epoch": 0.586, |
| "grad_norm": 0.1577884554862976, |
| "learning_rate": 0.0007997067970290741, |
| "loss": 3.3034, |
| "step": 29300 |
| }, |
| { |
| "epoch": 0.588, |
| "grad_norm": 0.1637413501739502, |
| "learning_rate": 0.0007932313693757703, |
| "loss": 3.3052, |
| "step": 29400 |
| }, |
| { |
| "epoch": 0.59, |
| "grad_norm": 0.19296114146709442, |
| "learning_rate": 0.0007867649864522074, |
| "loss": 3.3069, |
| "step": 29500 |
| }, |
| { |
| "epoch": 0.592, |
| "grad_norm": 0.14899814128875732, |
| "learning_rate": 0.0007803079311189227, |
| "loss": 3.3026, |
| "step": 29600 |
| }, |
| { |
| "epoch": 0.594, |
| "grad_norm": 0.1329040378332138, |
| "learning_rate": 0.0007738604858284345, |
| "loss": 3.2944, |
| "step": 29700 |
| }, |
| { |
| "epoch": 0.596, |
| "grad_norm": 0.15727417171001434, |
| "learning_rate": 0.0007674229326128867, |
| "loss": 3.3166, |
| "step": 29800 |
| }, |
| { |
| "epoch": 0.598, |
| "grad_norm": 0.15141139924526215, |
| "learning_rate": 0.0007609955530717117, |
| "loss": 3.2907, |
| "step": 29900 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 0.13476119935512543, |
| "learning_rate": 0.0007545786283593116, |
| "loss": 3.2933, |
| "step": 30000 |
| }, |
| { |
| "epoch": 0.6, |
| "eval_accuracy": 0.38492172211350295, |
| "eval_loss": 3.2943880558013916, |
| "eval_runtime": 11.9455, |
| "eval_samples_per_second": 83.713, |
| "eval_steps_per_second": 0.67, |
| "step": 30000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 128, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|