| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.76, |
| "eval_steps": 1000, |
| "global_step": 38000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 0.8063191175460815, |
| "learning_rate": 7.920000000000001e-05, |
| "loss": 8.2537, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 0.6161019206047058, |
| "learning_rate": 0.00015920000000000002, |
| "loss": 6.5731, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 0.8347169160842896, |
| "learning_rate": 0.00023920000000000001, |
| "loss": 6.1639, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.9150793552398682, |
| "learning_rate": 0.0003192, |
| "loss": 5.8682, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 0.6698479652404785, |
| "learning_rate": 0.0003992, |
| "loss": 5.5708, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.5323518514633179, |
| "learning_rate": 0.00047920000000000005, |
| "loss": 5.3157, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 0.7472386956214905, |
| "learning_rate": 0.0005592, |
| "loss": 5.0907, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.6033297181129456, |
| "learning_rate": 0.0006392, |
| "loss": 4.8805, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 0.497113436460495, |
| "learning_rate": 0.0007191999999999999, |
| "loss": 4.7207, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.47698545455932617, |
| "learning_rate": 0.0007992, |
| "loss": 4.6068, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_accuracy": 0.2703913894324853, |
| "eval_loss": 4.5053935050964355, |
| "eval_runtime": 13.0992, |
| "eval_samples_per_second": 76.341, |
| "eval_steps_per_second": 0.611, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 0.44341349601745605, |
| "learning_rate": 0.0008792, |
| "loss": 4.4904, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.4383724331855774, |
| "learning_rate": 0.0009592000000000001, |
| "loss": 4.439, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 0.3894674777984619, |
| "learning_rate": 0.0010391999999999999, |
| "loss": 4.3735, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.41903820633888245, |
| "learning_rate": 0.0011192, |
| "loss": 4.3406, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 0.36601418256759644, |
| "learning_rate": 0.0011992, |
| "loss": 4.2879, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.4282030165195465, |
| "learning_rate": 0.0012791999999999999, |
| "loss": 4.2625, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 0.3431214690208435, |
| "learning_rate": 0.0013592, |
| "loss": 4.2316, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.3303460478782654, |
| "learning_rate": 0.0014392, |
| "loss": 4.1976, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 0.3030431866645813, |
| "learning_rate": 0.0015192, |
| "loss": 4.1494, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.3076256513595581, |
| "learning_rate": 0.0015992, |
| "loss": 4.1688, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.30658512720156555, |
| "eval_loss": 4.094621658325195, |
| "eval_runtime": 12.4723, |
| "eval_samples_per_second": 80.178, |
| "eval_steps_per_second": 0.641, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 0.2820931077003479, |
| "learning_rate": 0.0016792, |
| "loss": 4.5201, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.23724448680877686, |
| "learning_rate": 0.0017592, |
| "loss": 4.1333, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 0.2325439602136612, |
| "learning_rate": 0.0018392, |
| "loss": 4.1072, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.23736131191253662, |
| "learning_rate": 0.0019192, |
| "loss": 4.0682, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.2313733696937561, |
| "learning_rate": 0.0019992, |
| "loss": 4.0571, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.22318924963474274, |
| "learning_rate": 0.001999978563623903, |
| "loss": 4.0153, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 0.20275355875492096, |
| "learning_rate": 0.0019999133871331223, |
| "loss": 4.0163, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.21306632459163666, |
| "learning_rate": 0.0019998044711915177, |
| "loss": 3.9777, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 0.19690750539302826, |
| "learning_rate": 0.0019996518205634257, |
| "loss": 3.9603, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.20464760065078735, |
| "learning_rate": 0.0019994554419262797, |
| "loss": 3.9566, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.3226555772994129, |
| "eval_loss": 3.9112932682037354, |
| "eval_runtime": 11.7171, |
| "eval_samples_per_second": 85.345, |
| "eval_steps_per_second": 0.683, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 0.18151508271694183, |
| "learning_rate": 0.001999215343870317, |
| "loss": 3.9501, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 0.18748600780963898, |
| "learning_rate": 0.0019989315368982054, |
| "loss": 3.9228, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 0.17895972728729248, |
| "learning_rate": 0.0019986040334245797, |
| "loss": 3.9041, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 0.18976759910583496, |
| "learning_rate": 0.001998232847775504, |
| "loss": 3.9123, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 0.1871727555990219, |
| "learning_rate": 0.0019978179961878404, |
| "loss": 3.8846, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 0.17986708879470825, |
| "learning_rate": 0.001997359496808541, |
| "loss": 3.8794, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 0.17090782523155212, |
| "learning_rate": 0.001996857369693855, |
| "loss": 3.8593, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 0.19015273451805115, |
| "learning_rate": 0.0019963116368084486, |
| "loss": 3.8642, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 0.1871633529663086, |
| "learning_rate": 0.001995722322024446, |
| "loss": 3.8479, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 0.18049727380275726, |
| "learning_rate": 0.001995089451120385, |
| "loss": 3.8226, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.33194520547945205, |
| "eval_loss": 3.8037452697753906, |
| "eval_runtime": 12.6853, |
| "eval_samples_per_second": 78.832, |
| "eval_steps_per_second": 0.631, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.082, |
| "grad_norm": 0.14800746738910675, |
| "learning_rate": 0.0019944130517800893, |
| "loss": 3.8319, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.084, |
| "grad_norm": 0.14713813364505768, |
| "learning_rate": 0.001993693153591457, |
| "loss": 3.8319, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.086, |
| "grad_norm": 0.17019635438919067, |
| "learning_rate": 0.0019929297880451674, |
| "loss": 3.8095, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.088, |
| "grad_norm": 0.15243591368198395, |
| "learning_rate": 0.0019921229885333023, |
| "loss": 3.8003, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.09, |
| "grad_norm": 0.16134986281394958, |
| "learning_rate": 0.001991272790347886, |
| "loss": 3.814, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.092, |
| "grad_norm": 0.16883811354637146, |
| "learning_rate": 0.0019903792306793415, |
| "loss": 3.7822, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.094, |
| "grad_norm": 0.15718473494052887, |
| "learning_rate": 0.001989442348614863, |
| "loss": 3.7902, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.096, |
| "grad_norm": 0.14631035923957825, |
| "learning_rate": 0.0019884621851367075, |
| "loss": 3.7805, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.098, |
| "grad_norm": 0.14771369099617004, |
| "learning_rate": 0.001987438783120401, |
| "loss": 3.7778, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 0.1536298543214798, |
| "learning_rate": 0.001986372187332862, |
| "loss": 3.7848, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.1, |
| "eval_accuracy": 0.340252446183953, |
| "eval_loss": 3.7290549278259277, |
| "eval_runtime": 11.9375, |
| "eval_samples_per_second": 83.77, |
| "eval_steps_per_second": 0.67, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.102, |
| "grad_norm": 0.15110976994037628, |
| "learning_rate": 0.0019852624444304467, |
| "loss": 3.7502, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.104, |
| "grad_norm": 0.15862922370433807, |
| "learning_rate": 0.0019841096029569044, |
| "loss": 3.761, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.106, |
| "grad_norm": 0.13948658108711243, |
| "learning_rate": 0.0019829137133412556, |
| "loss": 3.7591, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.108, |
| "grad_norm": 0.148148313164711, |
| "learning_rate": 0.001981674827895587, |
| "loss": 3.7385, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.11, |
| "grad_norm": 0.13170774281024933, |
| "learning_rate": 0.00198039300081276, |
| "loss": 3.7424, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.112, |
| "grad_norm": 0.14352479577064514, |
| "learning_rate": 0.0019790682881640448, |
| "loss": 3.7423, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.114, |
| "grad_norm": 0.12865835428237915, |
| "learning_rate": 0.001977700747896664, |
| "loss": 3.7263, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.116, |
| "grad_norm": 0.15470698475837708, |
| "learning_rate": 0.001976290439831259, |
| "loss": 3.7297, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.118, |
| "grad_norm": 0.1353532075881958, |
| "learning_rate": 0.0019748374256592736, |
| "loss": 3.7218, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 0.14982788264751434, |
| "learning_rate": 0.0019733417689402543, |
| "loss": 3.7164, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.12, |
| "eval_accuracy": 0.34496673189823873, |
| "eval_loss": 3.6746184825897217, |
| "eval_runtime": 11.9904, |
| "eval_samples_per_second": 83.4, |
| "eval_steps_per_second": 0.667, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.122, |
| "grad_norm": 0.14931881427764893, |
| "learning_rate": 0.0019718035350990712, |
| "loss": 3.7169, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.124, |
| "grad_norm": 0.14251381158828735, |
| "learning_rate": 0.0019702227914230566, |
| "loss": 3.6929, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.126, |
| "grad_norm": 0.14251849055290222, |
| "learning_rate": 0.001968599607059059, |
| "loss": 3.7018, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 0.15696577727794647, |
| "learning_rate": 0.0019669340530104207, |
| "loss": 3.7034, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.13, |
| "grad_norm": 0.14737001061439514, |
| "learning_rate": 0.001965226202133872, |
| "loss": 3.6889, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.132, |
| "grad_norm": 0.13667502999305725, |
| "learning_rate": 0.0019634761291363427, |
| "loss": 3.6935, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.134, |
| "grad_norm": 0.13390734791755676, |
| "learning_rate": 0.0019616839105716954, |
| "loss": 3.7032, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.136, |
| "grad_norm": 0.14834214746952057, |
| "learning_rate": 0.0019598496248373755, |
| "loss": 3.6657, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.138, |
| "grad_norm": 0.1314767599105835, |
| "learning_rate": 0.001957973352170984, |
| "loss": 3.673, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 0.15161661803722382, |
| "learning_rate": 0.001956055174646765, |
| "loss": 3.6713, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.14, |
| "eval_accuracy": 0.3484422700587084, |
| "eval_loss": 3.6349546909332275, |
| "eval_runtime": 12.1265, |
| "eval_samples_per_second": 82.464, |
| "eval_steps_per_second": 0.66, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.142, |
| "grad_norm": 0.12710200250148773, |
| "learning_rate": 0.0019540951761720174, |
| "loss": 3.67, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.144, |
| "grad_norm": 0.18498213589191437, |
| "learning_rate": 0.0019520934424834247, |
| "loss": 3.675, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.146, |
| "grad_norm": 0.13697299361228943, |
| "learning_rate": 0.0019500500611433025, |
| "loss": 3.6497, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.148, |
| "grad_norm": 0.13290269672870636, |
| "learning_rate": 0.0019479651215357707, |
| "loss": 3.662, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 0.12744054198265076, |
| "learning_rate": 0.0019458387148628417, |
| "loss": 3.6645, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.152, |
| "grad_norm": 0.13697493076324463, |
| "learning_rate": 0.001943670934140432, |
| "loss": 3.6374, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.154, |
| "grad_norm": 0.14067181944847107, |
| "learning_rate": 0.0019414618741942936, |
| "loss": 3.6453, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.156, |
| "grad_norm": 0.1316055953502655, |
| "learning_rate": 0.0019392116316558638, |
| "loss": 3.6679, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.158, |
| "grad_norm": 0.1261526495218277, |
| "learning_rate": 0.001936920304958042, |
| "loss": 3.6343, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 0.13644501566886902, |
| "learning_rate": 0.0019345879943308804, |
| "loss": 3.6365, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.16, |
| "eval_accuracy": 0.35229158512720155, |
| "eval_loss": 3.5980424880981445, |
| "eval_runtime": 17.6098, |
| "eval_samples_per_second": 56.787, |
| "eval_steps_per_second": 0.454, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.162, |
| "grad_norm": 0.13911907374858856, |
| "learning_rate": 0.0019322148017972016, |
| "loss": 3.646, |
| "step": 8100 |
| }, |
| { |
| "epoch": 0.164, |
| "grad_norm": 0.1297680288553238, |
| "learning_rate": 0.001929800831168135, |
| "loss": 3.6412, |
| "step": 8200 |
| }, |
| { |
| "epoch": 0.166, |
| "grad_norm": 0.14803367853164673, |
| "learning_rate": 0.001927346188038576, |
| "loss": 3.6298, |
| "step": 8300 |
| }, |
| { |
| "epoch": 0.168, |
| "grad_norm": 0.1325535923242569, |
| "learning_rate": 0.0019248509797825672, |
| "loss": 3.6067, |
| "step": 8400 |
| }, |
| { |
| "epoch": 0.17, |
| "grad_norm": 0.16948860883712769, |
| "learning_rate": 0.0019223153155486009, |
| "loss": 3.6284, |
| "step": 8500 |
| }, |
| { |
| "epoch": 0.172, |
| "grad_norm": 0.12973898649215698, |
| "learning_rate": 0.0019197393062548454, |
| "loss": 3.627, |
| "step": 8600 |
| }, |
| { |
| "epoch": 0.174, |
| "grad_norm": 0.1576964110136032, |
| "learning_rate": 0.0019171230645842923, |
| "loss": 3.6027, |
| "step": 8700 |
| }, |
| { |
| "epoch": 0.176, |
| "grad_norm": 0.15394999086856842, |
| "learning_rate": 0.0019144667049798272, |
| "loss": 3.6098, |
| "step": 8800 |
| }, |
| { |
| "epoch": 0.178, |
| "grad_norm": 0.12594962120056152, |
| "learning_rate": 0.0019117703436392253, |
| "loss": 3.6221, |
| "step": 8900 |
| }, |
| { |
| "epoch": 0.18, |
| "grad_norm": 0.1420104056596756, |
| "learning_rate": 0.001909034098510066, |
| "loss": 3.5993, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.18, |
| "eval_accuracy": 0.3549628180039139, |
| "eval_loss": 3.570964813232422, |
| "eval_runtime": 12.4303, |
| "eval_samples_per_second": 80.449, |
| "eval_steps_per_second": 0.644, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.182, |
| "grad_norm": 0.12720872461795807, |
| "learning_rate": 0.001906258089284576, |
| "loss": 3.5913, |
| "step": 9100 |
| }, |
| { |
| "epoch": 0.184, |
| "grad_norm": 0.14890190958976746, |
| "learning_rate": 0.0019034424373943915, |
| "loss": 3.6113, |
| "step": 9200 |
| }, |
| { |
| "epoch": 0.186, |
| "grad_norm": 0.15023267269134521, |
| "learning_rate": 0.0019005872660052478, |
| "loss": 3.6011, |
| "step": 9300 |
| }, |
| { |
| "epoch": 0.188, |
| "grad_norm": 0.1637164056301117, |
| "learning_rate": 0.001897692700011591, |
| "loss": 3.5932, |
| "step": 9400 |
| }, |
| { |
| "epoch": 0.19, |
| "grad_norm": 0.13102982938289642, |
| "learning_rate": 0.0018947588660311143, |
| "loss": 3.5889, |
| "step": 9500 |
| }, |
| { |
| "epoch": 0.192, |
| "grad_norm": 0.16267850995063782, |
| "learning_rate": 0.0018917858923992211, |
| "loss": 3.5939, |
| "step": 9600 |
| }, |
| { |
| "epoch": 0.194, |
| "grad_norm": 0.12798982858657837, |
| "learning_rate": 0.0018887739091634085, |
| "loss": 3.5896, |
| "step": 9700 |
| }, |
| { |
| "epoch": 0.196, |
| "grad_norm": 0.13872383534908295, |
| "learning_rate": 0.0018857230480775807, |
| "loss": 3.574, |
| "step": 9800 |
| }, |
| { |
| "epoch": 0.198, |
| "grad_norm": 0.11993825435638428, |
| "learning_rate": 0.0018826334425962855, |
| "loss": 3.589, |
| "step": 9900 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 0.13074541091918945, |
| "learning_rate": 0.0018795052278688753, |
| "loss": 3.5828, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.2, |
| "eval_accuracy": 0.35816046966731896, |
| "eval_loss": 3.540745973587036, |
| "eval_runtime": 17.8387, |
| "eval_samples_per_second": 56.058, |
| "eval_steps_per_second": 0.448, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.202, |
| "grad_norm": 0.13834109902381897, |
| "learning_rate": 0.0018763385407335963, |
| "loss": 3.5627, |
| "step": 10100 |
| }, |
| { |
| "epoch": 0.204, |
| "grad_norm": 0.15819378197193146, |
| "learning_rate": 0.001873133519711602, |
| "loss": 3.5634, |
| "step": 10200 |
| }, |
| { |
| "epoch": 0.206, |
| "grad_norm": 0.1618553251028061, |
| "learning_rate": 0.0018698903050008956, |
| "loss": 3.5768, |
| "step": 10300 |
| }, |
| { |
| "epoch": 0.208, |
| "grad_norm": 0.14207501709461212, |
| "learning_rate": 0.0018666090384701947, |
| "loss": 3.5676, |
| "step": 10400 |
| }, |
| { |
| "epoch": 0.21, |
| "grad_norm": 0.13513191044330597, |
| "learning_rate": 0.001863289863652727, |
| "loss": 3.5653, |
| "step": 10500 |
| }, |
| { |
| "epoch": 0.212, |
| "grad_norm": 0.12555710971355438, |
| "learning_rate": 0.0018599329257399516, |
| "loss": 3.5577, |
| "step": 10600 |
| }, |
| { |
| "epoch": 0.214, |
| "grad_norm": 0.129461869597435, |
| "learning_rate": 0.0018565383715752083, |
| "loss": 3.576, |
| "step": 10700 |
| }, |
| { |
| "epoch": 0.216, |
| "grad_norm": 0.11782678216695786, |
| "learning_rate": 0.0018531063496472927, |
| "loss": 3.5695, |
| "step": 10800 |
| }, |
| { |
| "epoch": 0.218, |
| "grad_norm": 0.1391923427581787, |
| "learning_rate": 0.0018496370100839622, |
| "loss": 3.5478, |
| "step": 10900 |
| }, |
| { |
| "epoch": 0.22, |
| "grad_norm": 0.15454861521720886, |
| "learning_rate": 0.0018461305046453683, |
| "loss": 3.56, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.22, |
| "eval_accuracy": 0.3610430528375734, |
| "eval_loss": 3.5181522369384766, |
| "eval_runtime": 12.5872, |
| "eval_samples_per_second": 79.446, |
| "eval_steps_per_second": 0.636, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.222, |
| "grad_norm": 0.14800433814525604, |
| "learning_rate": 0.0018425869867174187, |
| "loss": 3.5559, |
| "step": 11100 |
| }, |
| { |
| "epoch": 0.224, |
| "grad_norm": 0.16677607595920563, |
| "learning_rate": 0.0018390066113050665, |
| "loss": 3.5393, |
| "step": 11200 |
| }, |
| { |
| "epoch": 0.226, |
| "grad_norm": 0.17119687795639038, |
| "learning_rate": 0.0018353895350255317, |
| "loss": 3.5503, |
| "step": 11300 |
| }, |
| { |
| "epoch": 0.228, |
| "grad_norm": 0.16998586058616638, |
| "learning_rate": 0.0018317359161014477, |
| "loss": 3.5546, |
| "step": 11400 |
| }, |
| { |
| "epoch": 0.23, |
| "grad_norm": 0.16900590062141418, |
| "learning_rate": 0.001828045914353943, |
| "loss": 3.5485, |
| "step": 11500 |
| }, |
| { |
| "epoch": 0.232, |
| "grad_norm": 0.13857169449329376, |
| "learning_rate": 0.0018243196911956476, |
| "loss": 3.5357, |
| "step": 11600 |
| }, |
| { |
| "epoch": 0.234, |
| "grad_norm": 0.134566068649292, |
| "learning_rate": 0.0018205574096236336, |
| "loss": 3.5478, |
| "step": 11700 |
| }, |
| { |
| "epoch": 0.236, |
| "grad_norm": 0.13580797612667084, |
| "learning_rate": 0.0018167592342122857, |
| "loss": 3.5452, |
| "step": 11800 |
| }, |
| { |
| "epoch": 0.238, |
| "grad_norm": 0.16239416599273682, |
| "learning_rate": 0.0018129253311061002, |
| "loss": 3.5256, |
| "step": 11900 |
| }, |
| { |
| "epoch": 0.24, |
| "grad_norm": 0.16310077905654907, |
| "learning_rate": 0.0018090558680124193, |
| "loss": 3.5331, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.24, |
| "eval_accuracy": 0.3630156555772994, |
| "eval_loss": 3.4982736110687256, |
| "eval_runtime": 12.3123, |
| "eval_samples_per_second": 81.22, |
| "eval_steps_per_second": 0.65, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.242, |
| "grad_norm": 0.1448379009962082, |
| "learning_rate": 0.0018051510141940939, |
| "loss": 3.5416, |
| "step": 12100 |
| }, |
| { |
| "epoch": 0.244, |
| "grad_norm": 0.13583365082740784, |
| "learning_rate": 0.00180121094046208, |
| "loss": 3.5343, |
| "step": 12200 |
| }, |
| { |
| "epoch": 0.246, |
| "grad_norm": 0.17297838628292084, |
| "learning_rate": 0.001797235819167967, |
| "loss": 3.521, |
| "step": 12300 |
| }, |
| { |
| "epoch": 0.248, |
| "grad_norm": 0.15616372227668762, |
| "learning_rate": 0.0017932258241964375, |
| "loss": 3.533, |
| "step": 12400 |
| }, |
| { |
| "epoch": 0.25, |
| "grad_norm": 0.15943922102451324, |
| "learning_rate": 0.0017891811309576622, |
| "loss": 3.5177, |
| "step": 12500 |
| }, |
| { |
| "epoch": 0.252, |
| "grad_norm": 0.15389597415924072, |
| "learning_rate": 0.001785101916379627, |
| "loss": 3.5209, |
| "step": 12600 |
| }, |
| { |
| "epoch": 0.254, |
| "grad_norm": 0.13968117535114288, |
| "learning_rate": 0.0017809883589003912, |
| "loss": 3.5137, |
| "step": 12700 |
| }, |
| { |
| "epoch": 0.256, |
| "grad_norm": 0.1350937783718109, |
| "learning_rate": 0.0017768406384602862, |
| "loss": 3.5294, |
| "step": 12800 |
| }, |
| { |
| "epoch": 0.258, |
| "grad_norm": 0.13691306114196777, |
| "learning_rate": 0.00177265893649404, |
| "loss": 3.5228, |
| "step": 12900 |
| }, |
| { |
| "epoch": 0.26, |
| "grad_norm": 0.1496269851922989, |
| "learning_rate": 0.0017684434359228438, |
| "loss": 3.5026, |
| "step": 13000 |
| }, |
| { |
| "epoch": 0.26, |
| "eval_accuracy": 0.3645440313111546, |
| "eval_loss": 3.480961799621582, |
| "eval_runtime": 11.914, |
| "eval_samples_per_second": 83.935, |
| "eval_steps_per_second": 0.671, |
| "step": 13000 |
| }, |
| { |
| "epoch": 0.262, |
| "grad_norm": 0.1472535878419876, |
| "learning_rate": 0.0017641943211463488, |
| "loss": 3.5107, |
| "step": 13100 |
| }, |
| { |
| "epoch": 0.264, |
| "grad_norm": 0.1573815494775772, |
| "learning_rate": 0.0017599117780346006, |
| "loss": 3.5147, |
| "step": 13200 |
| }, |
| { |
| "epoch": 0.266, |
| "grad_norm": 0.15841859579086304, |
| "learning_rate": 0.0017555959939199084, |
| "loss": 3.501, |
| "step": 13300 |
| }, |
| { |
| "epoch": 0.268, |
| "grad_norm": 0.15096184611320496, |
| "learning_rate": 0.0017512471575886505, |
| "loss": 3.5034, |
| "step": 13400 |
| }, |
| { |
| "epoch": 0.27, |
| "grad_norm": 0.17669688165187836, |
| "learning_rate": 0.0017468654592730168, |
| "loss": 3.5101, |
| "step": 13500 |
| }, |
| { |
| "epoch": 0.272, |
| "grad_norm": 0.16740451753139496, |
| "learning_rate": 0.0017424510906426853, |
| "loss": 3.5021, |
| "step": 13600 |
| }, |
| { |
| "epoch": 0.274, |
| "grad_norm": 0.1432366967201233, |
| "learning_rate": 0.0017380042447964414, |
| "loss": 3.4915, |
| "step": 13700 |
| }, |
| { |
| "epoch": 0.276, |
| "grad_norm": 0.14669281244277954, |
| "learning_rate": 0.0017335251162537279, |
| "loss": 3.4975, |
| "step": 13800 |
| }, |
| { |
| "epoch": 0.278, |
| "grad_norm": 0.15456849336624146, |
| "learning_rate": 0.0017290139009461377, |
| "loss": 3.504, |
| "step": 13900 |
| }, |
| { |
| "epoch": 0.28, |
| "grad_norm": 0.163591668009758, |
| "learning_rate": 0.0017244707962088424, |
| "loss": 3.4899, |
| "step": 14000 |
| }, |
| { |
| "epoch": 0.28, |
| "eval_accuracy": 0.366587084148728, |
| "eval_loss": 3.4621543884277344, |
| "eval_runtime": 13.6803, |
| "eval_samples_per_second": 73.098, |
| "eval_steps_per_second": 0.585, |
| "step": 14000 |
| }, |
| { |
| "epoch": 0.282, |
| "grad_norm": 0.1512901335954666, |
| "learning_rate": 0.0017198960007719611, |
| "loss": 3.4777, |
| "step": 14100 |
| }, |
| { |
| "epoch": 0.284, |
| "grad_norm": 0.13871945440769196, |
| "learning_rate": 0.0017152897147518665, |
| "loss": 3.5067, |
| "step": 14200 |
| }, |
| { |
| "epoch": 0.286, |
| "grad_norm": 0.15246713161468506, |
| "learning_rate": 0.001710652139642431, |
| "loss": 3.5018, |
| "step": 14300 |
| }, |
| { |
| "epoch": 0.288, |
| "grad_norm": 0.13609080016613007, |
| "learning_rate": 0.0017059834783062142, |
| "loss": 3.4762, |
| "step": 14400 |
| }, |
| { |
| "epoch": 0.29, |
| "grad_norm": 0.15287351608276367, |
| "learning_rate": 0.0017012839349655868, |
| "loss": 3.4872, |
| "step": 14500 |
| }, |
| { |
| "epoch": 0.292, |
| "grad_norm": 0.14040616154670715, |
| "learning_rate": 0.001696553715193799, |
| "loss": 3.5042, |
| "step": 14600 |
| }, |
| { |
| "epoch": 0.294, |
| "grad_norm": 0.1571398377418518, |
| "learning_rate": 0.0016917930259059879, |
| "loss": 3.4744, |
| "step": 14700 |
| }, |
| { |
| "epoch": 0.296, |
| "grad_norm": 0.1505391150712967, |
| "learning_rate": 0.001687002075350125, |
| "loss": 3.4749, |
| "step": 14800 |
| }, |
| { |
| "epoch": 0.298, |
| "grad_norm": 0.14757487177848816, |
| "learning_rate": 0.001682181073097908, |
| "loss": 3.4852, |
| "step": 14900 |
| }, |
| { |
| "epoch": 0.3, |
| "grad_norm": 0.15271714329719543, |
| "learning_rate": 0.0016773302300355935, |
| "loss": 3.486, |
| "step": 15000 |
| }, |
| { |
| "epoch": 0.3, |
| "eval_accuracy": 0.36803326810176124, |
| "eval_loss": 3.4481563568115234, |
| "eval_runtime": 12.0908, |
| "eval_samples_per_second": 82.707, |
| "eval_steps_per_second": 0.662, |
| "step": 15000 |
| }, |
| { |
| "epoch": 0.302, |
| "grad_norm": 0.15327778458595276, |
| "learning_rate": 0.0016724497583547715, |
| "loss": 3.478, |
| "step": 15100 |
| }, |
| { |
| "epoch": 0.304, |
| "grad_norm": 0.16108167171478271, |
| "learning_rate": 0.0016675398715430842, |
| "loss": 3.4579, |
| "step": 15200 |
| }, |
| { |
| "epoch": 0.306, |
| "grad_norm": 0.15231835842132568, |
| "learning_rate": 0.001662600784374886, |
| "loss": 3.4819, |
| "step": 15300 |
| }, |
| { |
| "epoch": 0.308, |
| "grad_norm": 0.1505262553691864, |
| "learning_rate": 0.0016576327129018504, |
| "loss": 3.4808, |
| "step": 15400 |
| }, |
| { |
| "epoch": 0.31, |
| "grad_norm": 0.15296302735805511, |
| "learning_rate": 0.0016526358744435184, |
| "loss": 3.459, |
| "step": 15500 |
| }, |
| { |
| "epoch": 0.312, |
| "grad_norm": 0.2003161907196045, |
| "learning_rate": 0.001647610487577791, |
| "loss": 3.4751, |
| "step": 15600 |
| }, |
| { |
| "epoch": 0.314, |
| "grad_norm": 0.130265474319458, |
| "learning_rate": 0.0016425567721313694, |
| "loss": 3.4819, |
| "step": 15700 |
| }, |
| { |
| "epoch": 0.316, |
| "grad_norm": 0.14476525783538818, |
| "learning_rate": 0.0016374749491701401, |
| "loss": 3.4514, |
| "step": 15800 |
| }, |
| { |
| "epoch": 0.318, |
| "grad_norm": 0.1549970805644989, |
| "learning_rate": 0.0016323652409895013, |
| "loss": 3.4684, |
| "step": 15900 |
| }, |
| { |
| "epoch": 0.32, |
| "grad_norm": 0.14814923703670502, |
| "learning_rate": 0.0016272278711046424, |
| "loss": 3.4674, |
| "step": 16000 |
| }, |
| { |
| "epoch": 0.32, |
| "eval_accuracy": 0.36914090019569473, |
| "eval_loss": 3.435351848602295, |
| "eval_runtime": 12.375, |
| "eval_samples_per_second": 80.808, |
| "eval_steps_per_second": 0.646, |
| "step": 16000 |
| }, |
| { |
| "epoch": 0.322, |
| "grad_norm": 0.207662433385849, |
| "learning_rate": 0.001622063064240765, |
| "loss": 3.4661, |
| "step": 16100 |
| }, |
| { |
| "epoch": 0.324, |
| "grad_norm": 0.14697088301181793, |
| "learning_rate": 0.001616871046323253, |
| "loss": 3.4735, |
| "step": 16200 |
| }, |
| { |
| "epoch": 0.326, |
| "grad_norm": 0.16137182712554932, |
| "learning_rate": 0.0016116520444677894, |
| "loss": 3.451, |
| "step": 16300 |
| }, |
| { |
| "epoch": 0.328, |
| "grad_norm": 0.1839095950126648, |
| "learning_rate": 0.001606406286970423, |
| "loss": 3.4646, |
| "step": 16400 |
| }, |
| { |
| "epoch": 0.33, |
| "grad_norm": 0.1534615308046341, |
| "learning_rate": 0.0016011340032975805, |
| "loss": 3.4666, |
| "step": 16500 |
| }, |
| { |
| "epoch": 0.332, |
| "grad_norm": 0.15798258781433105, |
| "learning_rate": 0.00159583542407603, |
| "loss": 3.4418, |
| "step": 16600 |
| }, |
| { |
| "epoch": 0.334, |
| "grad_norm": 0.13258863985538483, |
| "learning_rate": 0.001590510781082791, |
| "loss": 3.4641, |
| "step": 16700 |
| }, |
| { |
| "epoch": 0.336, |
| "grad_norm": 0.14669552445411682, |
| "learning_rate": 0.001585160307234998, |
| "loss": 3.4667, |
| "step": 16800 |
| }, |
| { |
| "epoch": 0.338, |
| "grad_norm": 0.15622150897979736, |
| "learning_rate": 0.00157978423657971, |
| "loss": 3.439, |
| "step": 16900 |
| }, |
| { |
| "epoch": 0.34, |
| "grad_norm": 0.18012835085391998, |
| "learning_rate": 0.0015743828042836733, |
| "loss": 3.4512, |
| "step": 17000 |
| }, |
| { |
| "epoch": 0.34, |
| "eval_accuracy": 0.3706555772994129, |
| "eval_loss": 3.4220447540283203, |
| "eval_runtime": 21.6548, |
| "eval_samples_per_second": 46.179, |
| "eval_steps_per_second": 0.369, |
| "step": 17000 |
| }, |
| { |
| "epoch": 0.342, |
| "grad_norm": 0.143183171749115, |
| "learning_rate": 0.0015689562466230352, |
| "loss": 3.4531, |
| "step": 17100 |
| }, |
| { |
| "epoch": 0.344, |
| "grad_norm": 0.15830625593662262, |
| "learning_rate": 0.0015635048009730072, |
| "loss": 3.4425, |
| "step": 17200 |
| }, |
| { |
| "epoch": 0.346, |
| "grad_norm": 0.1351785659790039, |
| "learning_rate": 0.0015580287057974822, |
| "loss": 3.4491, |
| "step": 17300 |
| }, |
| { |
| "epoch": 0.348, |
| "grad_norm": 0.1572723239660263, |
| "learning_rate": 0.0015525282006386032, |
| "loss": 3.4329, |
| "step": 17400 |
| }, |
| { |
| "epoch": 0.35, |
| "grad_norm": 0.16905032098293304, |
| "learning_rate": 0.0015470035261062852, |
| "loss": 3.4457, |
| "step": 17500 |
| }, |
| { |
| "epoch": 0.352, |
| "grad_norm": 0.1542733907699585, |
| "learning_rate": 0.0015414549238676899, |
| "loss": 3.4466, |
| "step": 17600 |
| }, |
| { |
| "epoch": 0.354, |
| "grad_norm": 0.17319922149181366, |
| "learning_rate": 0.0015358826366366532, |
| "loss": 3.4311, |
| "step": 17700 |
| }, |
| { |
| "epoch": 0.356, |
| "grad_norm": 0.17198143899440765, |
| "learning_rate": 0.0015302869081630717, |
| "loss": 3.4453, |
| "step": 17800 |
| }, |
| { |
| "epoch": 0.358, |
| "grad_norm": 0.15461337566375732, |
| "learning_rate": 0.0015246679832222353, |
| "loss": 3.4539, |
| "step": 17900 |
| }, |
| { |
| "epoch": 0.36, |
| "grad_norm": 0.15917198359966278, |
| "learning_rate": 0.0015190261076041245, |
| "loss": 3.4188, |
| "step": 18000 |
| }, |
| { |
| "epoch": 0.36, |
| "eval_accuracy": 0.3723365949119374, |
| "eval_loss": 3.4081637859344482, |
| "eval_runtime": 12.6994, |
| "eval_samples_per_second": 78.744, |
| "eval_steps_per_second": 0.63, |
| "step": 18000 |
| }, |
| { |
| "epoch": 0.362, |
| "grad_norm": 0.15682971477508545, |
| "learning_rate": 0.0015133615281026555, |
| "loss": 3.426, |
| "step": 18100 |
| }, |
| { |
| "epoch": 0.364, |
| "grad_norm": 0.15073339641094208, |
| "learning_rate": 0.0015076744925048868, |
| "loss": 3.4358, |
| "step": 18200 |
| }, |
| { |
| "epoch": 0.366, |
| "grad_norm": 0.18040859699249268, |
| "learning_rate": 0.0015019652495801786, |
| "loss": 3.4261, |
| "step": 18300 |
| }, |
| { |
| "epoch": 0.368, |
| "grad_norm": 0.18911828100681305, |
| "learning_rate": 0.0014962340490693121, |
| "loss": 3.4359, |
| "step": 18400 |
| }, |
| { |
| "epoch": 0.37, |
| "grad_norm": 0.15816016495227814, |
| "learning_rate": 0.0014904811416735636, |
| "loss": 3.4216, |
| "step": 18500 |
| }, |
| { |
| "epoch": 0.372, |
| "grad_norm": 0.17654721438884735, |
| "learning_rate": 0.0014847067790437396, |
| "loss": 3.4347, |
| "step": 18600 |
| }, |
| { |
| "epoch": 0.374, |
| "grad_norm": 0.17527590692043304, |
| "learning_rate": 0.0014789112137691682, |
| "loss": 3.4256, |
| "step": 18700 |
| }, |
| { |
| "epoch": 0.376, |
| "grad_norm": 0.15412954986095428, |
| "learning_rate": 0.001473094699366649, |
| "loss": 3.4074, |
| "step": 18800 |
| }, |
| { |
| "epoch": 0.378, |
| "grad_norm": 0.1891617327928543, |
| "learning_rate": 0.0014672574902693661, |
| "loss": 3.4338, |
| "step": 18900 |
| }, |
| { |
| "epoch": 0.38, |
| "grad_norm": 0.16060151159763336, |
| "learning_rate": 0.001461399841815754, |
| "loss": 3.4372, |
| "step": 19000 |
| }, |
| { |
| "epoch": 0.38, |
| "eval_accuracy": 0.3734951076320939, |
| "eval_loss": 3.3982865810394287, |
| "eval_runtime": 12.1716, |
| "eval_samples_per_second": 82.158, |
| "eval_steps_per_second": 0.657, |
| "step": 19000 |
| }, |
| { |
| "epoch": 0.382, |
| "grad_norm": 0.17226573824882507, |
| "learning_rate": 0.0014555220102383326, |
| "loss": 3.4123, |
| "step": 19100 |
| }, |
| { |
| "epoch": 0.384, |
| "grad_norm": 0.15569022297859192, |
| "learning_rate": 0.0014496242526524964, |
| "loss": 3.4226, |
| "step": 19200 |
| }, |
| { |
| "epoch": 0.386, |
| "grad_norm": 0.16819970309734344, |
| "learning_rate": 0.001443706827045269, |
| "loss": 3.4294, |
| "step": 19300 |
| }, |
| { |
| "epoch": 0.388, |
| "grad_norm": 0.14779409766197205, |
| "learning_rate": 0.0014377699922640157, |
| "loss": 3.4233, |
| "step": 19400 |
| }, |
| { |
| "epoch": 0.39, |
| "grad_norm": 0.1541208177804947, |
| "learning_rate": 0.0014318140080051228, |
| "loss": 3.4087, |
| "step": 19500 |
| }, |
| { |
| "epoch": 0.392, |
| "grad_norm": 0.1646755486726761, |
| "learning_rate": 0.0014258391348026362, |
| "loss": 3.398, |
| "step": 19600 |
| }, |
| { |
| "epoch": 0.394, |
| "grad_norm": 0.18935076892375946, |
| "learning_rate": 0.0014198456340168662, |
| "loss": 3.4231, |
| "step": 19700 |
| }, |
| { |
| "epoch": 0.396, |
| "grad_norm": 0.1867913156747818, |
| "learning_rate": 0.0014138337678229528, |
| "loss": 3.4129, |
| "step": 19800 |
| }, |
| { |
| "epoch": 0.398, |
| "grad_norm": 0.15837714076042175, |
| "learning_rate": 0.0014078037991993992, |
| "loss": 3.3952, |
| "step": 19900 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 0.15315821766853333, |
| "learning_rate": 0.0014017559919165675, |
| "loss": 3.4155, |
| "step": 20000 |
| }, |
| { |
| "epoch": 0.4, |
| "eval_accuracy": 0.37439138943248534, |
| "eval_loss": 3.387746572494507, |
| "eval_runtime": 12.7148, |
| "eval_samples_per_second": 78.648, |
| "eval_steps_per_second": 0.629, |
| "step": 20000 |
| }, |
| { |
| "epoch": 0.402, |
| "grad_norm": 0.18359500169754028, |
| "learning_rate": 0.0013956906105251406, |
| "loss": 3.4094, |
| "step": 20100 |
| }, |
| { |
| "epoch": 0.404, |
| "grad_norm": 0.1576140820980072, |
| "learning_rate": 0.0013896079203445497, |
| "loss": 3.3935, |
| "step": 20200 |
| }, |
| { |
| "epoch": 0.406, |
| "grad_norm": 0.22549428045749664, |
| "learning_rate": 0.0013835081874513679, |
| "loss": 3.4006, |
| "step": 20300 |
| }, |
| { |
| "epoch": 0.408, |
| "grad_norm": 0.15331390500068665, |
| "learning_rate": 0.001377391678667673, |
| "loss": 3.4129, |
| "step": 20400 |
| }, |
| { |
| "epoch": 0.41, |
| "grad_norm": 0.16119052469730377, |
| "learning_rate": 0.0013712586615493736, |
| "loss": 3.4036, |
| "step": 20500 |
| }, |
| { |
| "epoch": 0.412, |
| "grad_norm": 0.15083439648151398, |
| "learning_rate": 0.0013651094043745067, |
| "loss": 3.3857, |
| "step": 20600 |
| }, |
| { |
| "epoch": 0.414, |
| "grad_norm": 0.14811307191848755, |
| "learning_rate": 0.0013589441761315021, |
| "loss": 3.4058, |
| "step": 20700 |
| }, |
| { |
| "epoch": 0.416, |
| "grad_norm": 0.15205927193164825, |
| "learning_rate": 0.0013527632465074157, |
| "loss": 3.4109, |
| "step": 20800 |
| }, |
| { |
| "epoch": 0.418, |
| "grad_norm": 0.164772629737854, |
| "learning_rate": 0.0013465668858761326, |
| "loss": 3.3811, |
| "step": 20900 |
| }, |
| { |
| "epoch": 0.42, |
| "grad_norm": 0.15865328907966614, |
| "learning_rate": 0.00134035536528654, |
| "loss": 3.3878, |
| "step": 21000 |
| }, |
| { |
| "epoch": 0.42, |
| "eval_accuracy": 0.37497260273972605, |
| "eval_loss": 3.3792359828948975, |
| "eval_runtime": 12.1681, |
| "eval_samples_per_second": 82.182, |
| "eval_steps_per_second": 0.657, |
| "step": 21000 |
| }, |
| { |
| "epoch": 0.422, |
| "grad_norm": 0.18703462183475494, |
| "learning_rate": 0.0013341289564506717, |
| "loss": 3.401, |
| "step": 21100 |
| }, |
| { |
| "epoch": 0.424, |
| "grad_norm": 0.15354931354522705, |
| "learning_rate": 0.0013278879317318202, |
| "loss": 3.3947, |
| "step": 21200 |
| }, |
| { |
| "epoch": 0.426, |
| "grad_norm": 0.14713342487812042, |
| "learning_rate": 0.0013216325641326255, |
| "loss": 3.3774, |
| "step": 21300 |
| }, |
| { |
| "epoch": 0.428, |
| "grad_norm": 0.17548827826976776, |
| "learning_rate": 0.0013153631272831304, |
| "loss": 3.3794, |
| "step": 21400 |
| }, |
| { |
| "epoch": 0.43, |
| "grad_norm": 0.1897333562374115, |
| "learning_rate": 0.001309079895428813, |
| "loss": 3.3955, |
| "step": 21500 |
| }, |
| { |
| "epoch": 0.432, |
| "grad_norm": 0.1576170176267624, |
| "learning_rate": 0.0013027831434185898, |
| "loss": 3.3884, |
| "step": 21600 |
| }, |
| { |
| "epoch": 0.434, |
| "grad_norm": 0.18555521965026855, |
| "learning_rate": 0.0012964731466927916, |
| "loss": 3.3695, |
| "step": 21700 |
| }, |
| { |
| "epoch": 0.436, |
| "grad_norm": 0.15987864136695862, |
| "learning_rate": 0.0012901501812711176, |
| "loss": 3.3891, |
| "step": 21800 |
| }, |
| { |
| "epoch": 0.438, |
| "grad_norm": 0.16719584167003632, |
| "learning_rate": 0.0012838145237405588, |
| "loss": 3.4056, |
| "step": 21900 |
| }, |
| { |
| "epoch": 0.44, |
| "grad_norm": 0.19731645286083221, |
| "learning_rate": 0.0012774664512432996, |
| "loss": 3.3772, |
| "step": 22000 |
| }, |
| { |
| "epoch": 0.44, |
| "eval_accuracy": 0.3771624266144814, |
| "eval_loss": 3.3640236854553223, |
| "eval_runtime": 12.4106, |
| "eval_samples_per_second": 80.576, |
| "eval_steps_per_second": 0.645, |
| "step": 22000 |
| }, |
| { |
| "epoch": 0.442, |
| "grad_norm": 0.1659466177225113, |
| "learning_rate": 0.0012711062414645967, |
| "loss": 3.3785, |
| "step": 22100 |
| }, |
| { |
| "epoch": 0.444, |
| "grad_norm": 0.15995340049266815, |
| "learning_rate": 0.0012647341726206296, |
| "loss": 3.389, |
| "step": 22200 |
| }, |
| { |
| "epoch": 0.446, |
| "grad_norm": 0.17644192278385162, |
| "learning_rate": 0.0012583505234463322, |
| "loss": 3.3787, |
| "step": 22300 |
| }, |
| { |
| "epoch": 0.448, |
| "grad_norm": 0.170969158411026, |
| "learning_rate": 0.0012519555731831996, |
| "loss": 3.365, |
| "step": 22400 |
| }, |
| { |
| "epoch": 0.45, |
| "grad_norm": 0.1832742989063263, |
| "learning_rate": 0.0012455496015670732, |
| "loss": 3.3835, |
| "step": 22500 |
| }, |
| { |
| "epoch": 0.452, |
| "grad_norm": 0.16345185041427612, |
| "learning_rate": 0.001239132888815904, |
| "loss": 3.3824, |
| "step": 22600 |
| }, |
| { |
| "epoch": 0.454, |
| "grad_norm": 0.16044080257415771, |
| "learning_rate": 0.0012327057156174945, |
| "loss": 3.3691, |
| "step": 22700 |
| }, |
| { |
| "epoch": 0.456, |
| "grad_norm": 0.1500287652015686, |
| "learning_rate": 0.0012262683631172222, |
| "loss": 3.3775, |
| "step": 22800 |
| }, |
| { |
| "epoch": 0.458, |
| "grad_norm": 0.1463031768798828, |
| "learning_rate": 0.0012198211129057393, |
| "loss": 3.377, |
| "step": 22900 |
| }, |
| { |
| "epoch": 0.46, |
| "grad_norm": 0.15354429185390472, |
| "learning_rate": 0.0012133642470066562, |
| "loss": 3.3741, |
| "step": 23000 |
| }, |
| { |
| "epoch": 0.46, |
| "eval_accuracy": 0.3781017612524462, |
| "eval_loss": 3.3543314933776855, |
| "eval_runtime": 14.9495, |
| "eval_samples_per_second": 66.892, |
| "eval_steps_per_second": 0.535, |
| "step": 23000 |
| }, |
| { |
| "epoch": 0.462, |
| "grad_norm": 0.16891853511333466, |
| "learning_rate": 0.0012068980478642047, |
| "loss": 3.3573, |
| "step": 23100 |
| }, |
| { |
| "epoch": 0.464, |
| "grad_norm": 0.15053242444992065, |
| "learning_rate": 0.0012004227983308828, |
| "loss": 3.3767, |
| "step": 23200 |
| }, |
| { |
| "epoch": 0.466, |
| "grad_norm": 0.13770438730716705, |
| "learning_rate": 0.001193938781655082, |
| "loss": 3.3735, |
| "step": 23300 |
| }, |
| { |
| "epoch": 0.468, |
| "grad_norm": 0.16500090062618256, |
| "learning_rate": 0.0011874462814686975, |
| "loss": 3.3594, |
| "step": 23400 |
| }, |
| { |
| "epoch": 0.47, |
| "grad_norm": 0.15050701797008514, |
| "learning_rate": 0.0011809455817747194, |
| "loss": 3.3736, |
| "step": 23500 |
| }, |
| { |
| "epoch": 0.472, |
| "grad_norm": 0.14346493780612946, |
| "learning_rate": 0.0011744369669348122, |
| "loss": 3.3714, |
| "step": 23600 |
| }, |
| { |
| "epoch": 0.474, |
| "grad_norm": 0.15672604739665985, |
| "learning_rate": 0.0011679207216568738, |
| "loss": 3.3547, |
| "step": 23700 |
| }, |
| { |
| "epoch": 0.476, |
| "grad_norm": 0.20898281037807465, |
| "learning_rate": 0.0011613971309825828, |
| "loss": 3.3603, |
| "step": 23800 |
| }, |
| { |
| "epoch": 0.478, |
| "grad_norm": 0.15672914683818817, |
| "learning_rate": 0.001154866480274928, |
| "loss": 3.363, |
| "step": 23900 |
| }, |
| { |
| "epoch": 0.48, |
| "grad_norm": 0.15417250990867615, |
| "learning_rate": 0.0011483290552057285, |
| "loss": 3.3706, |
| "step": 24000 |
| }, |
| { |
| "epoch": 0.48, |
| "eval_accuracy": 0.37883365949119374, |
| "eval_loss": 3.344762086868286, |
| "eval_runtime": 12.8887, |
| "eval_samples_per_second": 77.587, |
| "eval_steps_per_second": 0.621, |
| "step": 24000 |
| }, |
| { |
| "epoch": 0.482, |
| "grad_norm": 0.17039372026920319, |
| "learning_rate": 0.0011417851417431346, |
| "loss": 3.3587, |
| "step": 24100 |
| }, |
| { |
| "epoch": 0.484, |
| "grad_norm": 0.16372039914131165, |
| "learning_rate": 0.0011352350261391204, |
| "loss": 3.3488, |
| "step": 24200 |
| }, |
| { |
| "epoch": 0.486, |
| "grad_norm": 0.15907979011535645, |
| "learning_rate": 0.0011286789949169628, |
| "loss": 3.3638, |
| "step": 24300 |
| }, |
| { |
| "epoch": 0.488, |
| "grad_norm": 0.15641772747039795, |
| "learning_rate": 0.001122117334858705, |
| "loss": 3.3627, |
| "step": 24400 |
| }, |
| { |
| "epoch": 0.49, |
| "grad_norm": 0.17920181155204773, |
| "learning_rate": 0.0011155503329926151, |
| "loss": 3.3278, |
| "step": 24500 |
| }, |
| { |
| "epoch": 0.492, |
| "grad_norm": 0.16552342474460602, |
| "learning_rate": 0.001108978276580629, |
| "loss": 3.3653, |
| "step": 24600 |
| }, |
| { |
| "epoch": 0.494, |
| "grad_norm": 0.17901502549648285, |
| "learning_rate": 0.001102401453105785, |
| "loss": 3.3626, |
| "step": 24700 |
| }, |
| { |
| "epoch": 0.496, |
| "grad_norm": 0.17022928595542908, |
| "learning_rate": 0.001095820150259647, |
| "loss": 3.3391, |
| "step": 24800 |
| }, |
| { |
| "epoch": 0.498, |
| "grad_norm": 0.1806282252073288, |
| "learning_rate": 0.0010892346559297226, |
| "loss": 3.3476, |
| "step": 24900 |
| }, |
| { |
| "epoch": 0.5, |
| "grad_norm": 0.16578249633312225, |
| "learning_rate": 0.0010826452581868676, |
| "loss": 3.3586, |
| "step": 25000 |
| }, |
| { |
| "epoch": 0.5, |
| "eval_accuracy": 0.38070841487279844, |
| "eval_loss": 3.3385705947875977, |
| "eval_runtime": 17.5524, |
| "eval_samples_per_second": 56.972, |
| "eval_steps_per_second": 0.456, |
| "step": 25000 |
| }, |
| { |
| "epoch": 0.502, |
| "grad_norm": 0.14317767322063446, |
| "learning_rate": 0.001076052245272686, |
| "loss": 3.3494, |
| "step": 25100 |
| }, |
| { |
| "epoch": 0.504, |
| "grad_norm": 0.18083012104034424, |
| "learning_rate": 0.0010694559055869205, |
| "loss": 3.3425, |
| "step": 25200 |
| }, |
| { |
| "epoch": 0.506, |
| "grad_norm": 0.1475485861301422, |
| "learning_rate": 0.0010628565276748394, |
| "loss": 3.3346, |
| "step": 25300 |
| }, |
| { |
| "epoch": 0.508, |
| "grad_norm": 0.19806760549545288, |
| "learning_rate": 0.0010562544002146108, |
| "loss": 3.359, |
| "step": 25400 |
| }, |
| { |
| "epoch": 0.51, |
| "grad_norm": 0.1636313647031784, |
| "learning_rate": 0.0010496498120046778, |
| "loss": 3.351, |
| "step": 25500 |
| }, |
| { |
| "epoch": 0.512, |
| "grad_norm": 0.14796622097492218, |
| "learning_rate": 0.0010430430519511244, |
| "loss": 3.3308, |
| "step": 25600 |
| }, |
| { |
| "epoch": 0.514, |
| "grad_norm": 0.15158209204673767, |
| "learning_rate": 0.001036434409055039, |
| "loss": 3.3462, |
| "step": 25700 |
| }, |
| { |
| "epoch": 0.516, |
| "grad_norm": 0.1509816199541092, |
| "learning_rate": 0.0010298241723998701, |
| "loss": 3.3612, |
| "step": 25800 |
| }, |
| { |
| "epoch": 0.518, |
| "grad_norm": 0.2039985954761505, |
| "learning_rate": 0.001023212631138783, |
| "loss": 3.3233, |
| "step": 25900 |
| }, |
| { |
| "epoch": 0.52, |
| "grad_norm": 0.15846829116344452, |
| "learning_rate": 0.001016600074482012, |
| "loss": 3.3338, |
| "step": 26000 |
| }, |
| { |
| "epoch": 0.52, |
| "eval_accuracy": 0.381559686888454, |
| "eval_loss": 3.330564498901367, |
| "eval_runtime": 11.5688, |
| "eval_samples_per_second": 86.439, |
| "eval_steps_per_second": 0.692, |
| "step": 26000 |
| }, |
| { |
| "epoch": 0.522, |
| "grad_norm": 0.16366560757160187, |
| "learning_rate": 0.0010099867916842063, |
| "loss": 3.349, |
| "step": 26100 |
| }, |
| { |
| "epoch": 0.524, |
| "grad_norm": 0.18745005130767822, |
| "learning_rate": 0.00100337307203178, |
| "loss": 3.3381, |
| "step": 26200 |
| }, |
| { |
| "epoch": 0.526, |
| "grad_norm": 0.14261376857757568, |
| "learning_rate": 0.0009967592048302552, |
| "loss": 3.334, |
| "step": 26300 |
| }, |
| { |
| "epoch": 0.528, |
| "grad_norm": 0.16403239965438843, |
| "learning_rate": 0.0009901454793916102, |
| "loss": 3.3253, |
| "step": 26400 |
| }, |
| { |
| "epoch": 0.53, |
| "grad_norm": 0.1447836309671402, |
| "learning_rate": 0.00098353218502162, |
| "loss": 3.3442, |
| "step": 26500 |
| }, |
| { |
| "epoch": 0.532, |
| "grad_norm": 0.18312039971351624, |
| "learning_rate": 0.0009769196110072061, |
| "loss": 3.338, |
| "step": 26600 |
| }, |
| { |
| "epoch": 0.534, |
| "grad_norm": 0.1727229803800583, |
| "learning_rate": 0.0009703080466037767, |
| "loss": 3.3183, |
| "step": 26700 |
| }, |
| { |
| "epoch": 0.536, |
| "grad_norm": 0.15190576016902924, |
| "learning_rate": 0.0009636977810225777, |
| "loss": 3.3387, |
| "step": 26800 |
| }, |
| { |
| "epoch": 0.538, |
| "grad_norm": 0.1397494077682495, |
| "learning_rate": 0.0009570891034180394, |
| "loss": 3.3415, |
| "step": 26900 |
| }, |
| { |
| "epoch": 0.54, |
| "grad_norm": 0.1609794944524765, |
| "learning_rate": 0.0009504823028751295, |
| "loss": 3.3174, |
| "step": 27000 |
| }, |
| { |
| "epoch": 0.54, |
| "eval_accuracy": 0.38231311154598824, |
| "eval_loss": 3.3193719387054443, |
| "eval_runtime": 11.9776, |
| "eval_samples_per_second": 83.489, |
| "eval_steps_per_second": 0.668, |
| "step": 27000 |
| }, |
| { |
| "epoch": 0.542, |
| "grad_norm": 0.14202991127967834, |
| "learning_rate": 0.0009438776683967071, |
| "loss": 3.334, |
| "step": 27100 |
| }, |
| { |
| "epoch": 0.544, |
| "grad_norm": 0.14222672581672668, |
| "learning_rate": 0.0009372754888908796, |
| "loss": 3.329, |
| "step": 27200 |
| }, |
| { |
| "epoch": 0.546, |
| "grad_norm": 0.2155715525150299, |
| "learning_rate": 0.000930676053158368, |
| "loss": 3.3328, |
| "step": 27300 |
| }, |
| { |
| "epoch": 0.548, |
| "grad_norm": 0.14550994336605072, |
| "learning_rate": 0.0009240796498798694, |
| "loss": 3.3314, |
| "step": 27400 |
| }, |
| { |
| "epoch": 0.55, |
| "grad_norm": 0.1446721851825714, |
| "learning_rate": 0.0009174865676034329, |
| "loss": 3.322, |
| "step": 27500 |
| }, |
| { |
| "epoch": 0.552, |
| "grad_norm": 0.15072402358055115, |
| "learning_rate": 0.000910897094731836, |
| "loss": 3.3292, |
| "step": 27600 |
| }, |
| { |
| "epoch": 0.554, |
| "grad_norm": 0.14199484884738922, |
| "learning_rate": 0.0009043115195099688, |
| "loss": 3.3319, |
| "step": 27700 |
| }, |
| { |
| "epoch": 0.556, |
| "grad_norm": 0.1559506356716156, |
| "learning_rate": 0.0008977301300122259, |
| "loss": 3.3161, |
| "step": 27800 |
| }, |
| { |
| "epoch": 0.558, |
| "grad_norm": 0.14755868911743164, |
| "learning_rate": 0.0008911532141299046, |
| "loss": 3.3283, |
| "step": 27900 |
| }, |
| { |
| "epoch": 0.56, |
| "grad_norm": 0.17486320436000824, |
| "learning_rate": 0.000884581059558612, |
| "loss": 3.3317, |
| "step": 28000 |
| }, |
| { |
| "epoch": 0.56, |
| "eval_accuracy": 0.38318786692759293, |
| "eval_loss": 3.3127212524414062, |
| "eval_runtime": 13.6461, |
| "eval_samples_per_second": 73.281, |
| "eval_steps_per_second": 0.586, |
| "step": 28000 |
| }, |
| { |
| "epoch": 0.562, |
| "grad_norm": 0.15159651637077332, |
| "learning_rate": 0.00087801395378568, |
| "loss": 3.3107, |
| "step": 28100 |
| }, |
| { |
| "epoch": 0.564, |
| "grad_norm": 0.16269639134407043, |
| "learning_rate": 0.0008714521840775893, |
| "loss": 3.3182, |
| "step": 28200 |
| }, |
| { |
| "epoch": 0.566, |
| "grad_norm": 0.16659656167030334, |
| "learning_rate": 0.0008648960374674042, |
| "loss": 3.3197, |
| "step": 28300 |
| }, |
| { |
| "epoch": 0.568, |
| "grad_norm": 0.1418466717004776, |
| "learning_rate": 0.0008583458007422165, |
| "loss": 3.3122, |
| "step": 28400 |
| }, |
| { |
| "epoch": 0.57, |
| "grad_norm": 0.14337651431560516, |
| "learning_rate": 0.0008518017604306002, |
| "loss": 3.3163, |
| "step": 28500 |
| }, |
| { |
| "epoch": 0.572, |
| "grad_norm": 0.16096735000610352, |
| "learning_rate": 0.0008452642027900783, |
| "loss": 3.3061, |
| "step": 28600 |
| }, |
| { |
| "epoch": 0.574, |
| "grad_norm": 0.18552157282829285, |
| "learning_rate": 0.0008387334137946008, |
| "loss": 3.318, |
| "step": 28700 |
| }, |
| { |
| "epoch": 0.576, |
| "grad_norm": 0.16036978363990784, |
| "learning_rate": 0.0008322096791220355, |
| "loss": 3.3109, |
| "step": 28800 |
| }, |
| { |
| "epoch": 0.578, |
| "grad_norm": 0.16824260354042053, |
| "learning_rate": 0.0008256932841416705, |
| "loss": 3.3074, |
| "step": 28900 |
| }, |
| { |
| "epoch": 0.58, |
| "grad_norm": 0.16867192089557648, |
| "learning_rate": 0.0008191845139017326, |
| "loss": 3.3225, |
| "step": 29000 |
| }, |
| { |
| "epoch": 0.58, |
| "eval_accuracy": 0.38399021526418786, |
| "eval_loss": 3.3025898933410645, |
| "eval_runtime": 11.9945, |
| "eval_samples_per_second": 83.372, |
| "eval_steps_per_second": 0.667, |
| "step": 29000 |
| }, |
| { |
| "epoch": 0.582, |
| "grad_norm": 0.1469780057668686, |
| "learning_rate": 0.0008126836531169177, |
| "loss": 3.3142, |
| "step": 29100 |
| }, |
| { |
| "epoch": 0.584, |
| "grad_norm": 0.15736377239227295, |
| "learning_rate": 0.0008061909861559362, |
| "loss": 3.2969, |
| "step": 29200 |
| }, |
| { |
| "epoch": 0.586, |
| "grad_norm": 0.1577884554862976, |
| "learning_rate": 0.0007997067970290741, |
| "loss": 3.3034, |
| "step": 29300 |
| }, |
| { |
| "epoch": 0.588, |
| "grad_norm": 0.1637413501739502, |
| "learning_rate": 0.0007932313693757703, |
| "loss": 3.3052, |
| "step": 29400 |
| }, |
| { |
| "epoch": 0.59, |
| "grad_norm": 0.19296114146709442, |
| "learning_rate": 0.0007867649864522074, |
| "loss": 3.3069, |
| "step": 29500 |
| }, |
| { |
| "epoch": 0.592, |
| "grad_norm": 0.14899814128875732, |
| "learning_rate": 0.0007803079311189227, |
| "loss": 3.3026, |
| "step": 29600 |
| }, |
| { |
| "epoch": 0.594, |
| "grad_norm": 0.1329040378332138, |
| "learning_rate": 0.0007738604858284345, |
| "loss": 3.2944, |
| "step": 29700 |
| }, |
| { |
| "epoch": 0.596, |
| "grad_norm": 0.15727417171001434, |
| "learning_rate": 0.0007674229326128867, |
| "loss": 3.3166, |
| "step": 29800 |
| }, |
| { |
| "epoch": 0.598, |
| "grad_norm": 0.15141139924526215, |
| "learning_rate": 0.0007609955530717117, |
| "loss": 3.2907, |
| "step": 29900 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 0.13476119935512543, |
| "learning_rate": 0.0007545786283593116, |
| "loss": 3.2933, |
| "step": 30000 |
| }, |
| { |
| "epoch": 0.6, |
| "eval_accuracy": 0.38492172211350295, |
| "eval_loss": 3.2943880558013916, |
| "eval_runtime": 11.9455, |
| "eval_samples_per_second": 83.713, |
| "eval_steps_per_second": 0.67, |
| "step": 30000 |
| }, |
| { |
| "epoch": 0.602, |
| "grad_norm": 0.19772815704345703, |
| "learning_rate": 0.0007481724391727628, |
| "loss": 3.3089, |
| "step": 30100 |
| }, |
| { |
| "epoch": 0.604, |
| "grad_norm": 0.16669701039791107, |
| "learning_rate": 0.0007417772657395325, |
| "loss": 3.3032, |
| "step": 30200 |
| }, |
| { |
| "epoch": 0.606, |
| "grad_norm": 0.15029844641685486, |
| "learning_rate": 0.0007353933878052245, |
| "loss": 3.2961, |
| "step": 30300 |
| }, |
| { |
| "epoch": 0.608, |
| "grad_norm": 0.1436147391796112, |
| "learning_rate": 0.0007290210846213408, |
| "loss": 3.2984, |
| "step": 30400 |
| }, |
| { |
| "epoch": 0.61, |
| "grad_norm": 0.1456470638513565, |
| "learning_rate": 0.0007226606349330653, |
| "loss": 3.3059, |
| "step": 30500 |
| }, |
| { |
| "epoch": 0.612, |
| "grad_norm": 0.15387193858623505, |
| "learning_rate": 0.0007163123169670729, |
| "loss": 3.3034, |
| "step": 30600 |
| }, |
| { |
| "epoch": 0.614, |
| "grad_norm": 0.1466827392578125, |
| "learning_rate": 0.0007099764084193565, |
| "loss": 3.278, |
| "step": 30700 |
| }, |
| { |
| "epoch": 0.616, |
| "grad_norm": 0.17281971871852875, |
| "learning_rate": 0.0007036531864430824, |
| "loss": 3.2818, |
| "step": 30800 |
| }, |
| { |
| "epoch": 0.618, |
| "grad_norm": 0.15874525904655457, |
| "learning_rate": 0.0006973429276364641, |
| "loss": 3.3064, |
| "step": 30900 |
| }, |
| { |
| "epoch": 0.62, |
| "grad_norm": 0.14938025176525116, |
| "learning_rate": 0.0006910459080306636, |
| "loss": 3.2793, |
| "step": 31000 |
| }, |
| { |
| "epoch": 0.62, |
| "eval_accuracy": 0.3857945205479452, |
| "eval_loss": 3.2896742820739746, |
| "eval_runtime": 12.6346, |
| "eval_samples_per_second": 79.148, |
| "eval_steps_per_second": 0.633, |
| "step": 31000 |
| }, |
| { |
| "epoch": 0.622, |
| "grad_norm": 0.12781454622745514, |
| "learning_rate": 0.0006847624030777184, |
| "loss": 3.2855, |
| "step": 31100 |
| }, |
| { |
| "epoch": 0.624, |
| "grad_norm": 0.15578420460224152, |
| "learning_rate": 0.0006784926876384907, |
| "loss": 3.2931, |
| "step": 31200 |
| }, |
| { |
| "epoch": 0.626, |
| "grad_norm": 0.1583343893289566, |
| "learning_rate": 0.000672237035970645, |
| "loss": 3.283, |
| "step": 31300 |
| }, |
| { |
| "epoch": 0.628, |
| "grad_norm": 0.14310809969902039, |
| "learning_rate": 0.0006659957217166504, |
| "loss": 3.2721, |
| "step": 31400 |
| }, |
| { |
| "epoch": 0.63, |
| "grad_norm": 0.19646517932415009, |
| "learning_rate": 0.0006597690178918121, |
| "loss": 3.2853, |
| "step": 31500 |
| }, |
| { |
| "epoch": 0.632, |
| "grad_norm": 0.1555543690919876, |
| "learning_rate": 0.0006535571968723266, |
| "loss": 3.2907, |
| "step": 31600 |
| }, |
| { |
| "epoch": 0.634, |
| "grad_norm": 0.15089106559753418, |
| "learning_rate": 0.0006473605303833691, |
| "loss": 3.2856, |
| "step": 31700 |
| }, |
| { |
| "epoch": 0.636, |
| "grad_norm": 0.15569235384464264, |
| "learning_rate": 0.000641179289487206, |
| "loss": 3.268, |
| "step": 31800 |
| }, |
| { |
| "epoch": 0.638, |
| "grad_norm": 0.15568938851356506, |
| "learning_rate": 0.0006350137445713386, |
| "loss": 3.2885, |
| "step": 31900 |
| }, |
| { |
| "epoch": 0.64, |
| "grad_norm": 0.14168080687522888, |
| "learning_rate": 0.0006288641653366752, |
| "loss": 3.2906, |
| "step": 32000 |
| }, |
| { |
| "epoch": 0.64, |
| "eval_accuracy": 0.38636986301369863, |
| "eval_loss": 3.280740976333618, |
| "eval_runtime": 12.2058, |
| "eval_samples_per_second": 81.928, |
| "eval_steps_per_second": 0.655, |
| "step": 32000 |
| }, |
| { |
| "epoch": 0.642, |
| "grad_norm": 0.15901386737823486, |
| "learning_rate": 0.0006227308207857332, |
| "loss": 3.2593, |
| "step": 32100 |
| }, |
| { |
| "epoch": 0.644, |
| "grad_norm": 0.20247270166873932, |
| "learning_rate": 0.0006166139792108727, |
| "loss": 3.2716, |
| "step": 32200 |
| }, |
| { |
| "epoch": 0.646, |
| "grad_norm": 0.1866617053747177, |
| "learning_rate": 0.0006105139081825601, |
| "loss": 3.2867, |
| "step": 32300 |
| }, |
| { |
| "epoch": 0.648, |
| "grad_norm": 0.14510272443294525, |
| "learning_rate": 0.0006044308745376636, |
| "loss": 3.273, |
| "step": 32400 |
| }, |
| { |
| "epoch": 0.65, |
| "grad_norm": 0.13237079977989197, |
| "learning_rate": 0.000598365144367781, |
| "loss": 3.2621, |
| "step": 32500 |
| }, |
| { |
| "epoch": 0.652, |
| "grad_norm": 0.15459007024765015, |
| "learning_rate": 0.0005923169830076003, |
| "loss": 3.2658, |
| "step": 32600 |
| }, |
| { |
| "epoch": 0.654, |
| "grad_norm": 0.13574740290641785, |
| "learning_rate": 0.0005862866550232925, |
| "loss": 3.2789, |
| "step": 32700 |
| }, |
| { |
| "epoch": 0.656, |
| "grad_norm": 0.15900640189647675, |
| "learning_rate": 0.0005802744242009394, |
| "loss": 3.2648, |
| "step": 32800 |
| }, |
| { |
| "epoch": 0.658, |
| "grad_norm": 0.18648633360862732, |
| "learning_rate": 0.000574280553534994, |
| "loss": 3.2575, |
| "step": 32900 |
| }, |
| { |
| "epoch": 0.66, |
| "grad_norm": 0.15465696156024933, |
| "learning_rate": 0.0005683053052167772, |
| "loss": 3.2787, |
| "step": 33000 |
| }, |
| { |
| "epoch": 0.66, |
| "eval_accuracy": 0.3878962818003914, |
| "eval_loss": 3.276221513748169, |
| "eval_runtime": 11.2808, |
| "eval_samples_per_second": 88.646, |
| "eval_steps_per_second": 0.709, |
| "step": 33000 |
| }, |
| { |
| "epoch": 0.662, |
| "grad_norm": 0.15960443019866943, |
| "learning_rate": 0.000562348940623007, |
| "loss": 3.2866, |
| "step": 33100 |
| }, |
| { |
| "epoch": 0.664, |
| "grad_norm": 0.13718590140342712, |
| "learning_rate": 0.0005564117203043664, |
| "loss": 3.2549, |
| "step": 33200 |
| }, |
| { |
| "epoch": 0.666, |
| "grad_norm": 0.1479761153459549, |
| "learning_rate": 0.0005504939039741067, |
| "loss": 3.2697, |
| "step": 33300 |
| }, |
| { |
| "epoch": 0.668, |
| "grad_norm": 0.15424597263336182, |
| "learning_rate": 0.0005445957504966848, |
| "loss": 3.2743, |
| "step": 33400 |
| }, |
| { |
| "epoch": 0.67, |
| "grad_norm": 0.13904234766960144, |
| "learning_rate": 0.0005387175178764416, |
| "loss": 3.2551, |
| "step": 33500 |
| }, |
| { |
| "epoch": 0.672, |
| "grad_norm": 0.18515262007713318, |
| "learning_rate": 0.000532859463246314, |
| "loss": 3.2581, |
| "step": 33600 |
| }, |
| { |
| "epoch": 0.674, |
| "grad_norm": 0.1791761815547943, |
| "learning_rate": 0.0005270218428565896, |
| "loss": 3.2701, |
| "step": 33700 |
| }, |
| { |
| "epoch": 0.676, |
| "grad_norm": 0.17315128445625305, |
| "learning_rate": 0.0005212049120636959, |
| "loss": 3.2631, |
| "step": 33800 |
| }, |
| { |
| "epoch": 0.678, |
| "grad_norm": 0.1509447544813156, |
| "learning_rate": 0.000515408925319029, |
| "loss": 3.2592, |
| "step": 33900 |
| }, |
| { |
| "epoch": 0.68, |
| "grad_norm": 0.1570684164762497, |
| "learning_rate": 0.0005096341361578265, |
| "loss": 3.2685, |
| "step": 34000 |
| }, |
| { |
| "epoch": 0.68, |
| "eval_accuracy": 0.3882113502935421, |
| "eval_loss": 3.2670066356658936, |
| "eval_runtime": 11.2814, |
| "eval_samples_per_second": 88.642, |
| "eval_steps_per_second": 0.709, |
| "step": 34000 |
| }, |
| { |
| "epoch": 0.682, |
| "grad_norm": 0.1495920717716217, |
| "learning_rate": 0.000503880797188073, |
| "loss": 3.2605, |
| "step": 34100 |
| }, |
| { |
| "epoch": 0.684, |
| "grad_norm": 0.16431903839111328, |
| "learning_rate": 0.0004981491600794544, |
| "loss": 3.2526, |
| "step": 34200 |
| }, |
| { |
| "epoch": 0.686, |
| "grad_norm": 0.1864980161190033, |
| "learning_rate": 0.0004924394755523447, |
| "loss": 3.2562, |
| "step": 34300 |
| }, |
| { |
| "epoch": 0.688, |
| "grad_norm": 0.1656457781791687, |
| "learning_rate": 0.0004867519933668425, |
| "loss": 3.2649, |
| "step": 34400 |
| }, |
| { |
| "epoch": 0.69, |
| "grad_norm": 0.165181502699852, |
| "learning_rate": 0.0004810869623118437, |
| "loss": 3.257, |
| "step": 34500 |
| }, |
| { |
| "epoch": 0.692, |
| "grad_norm": 0.13930366933345795, |
| "learning_rate": 0.0004754446301941582, |
| "loss": 3.2525, |
| "step": 34600 |
| }, |
| { |
| "epoch": 0.694, |
| "grad_norm": 0.1519293636083603, |
| "learning_rate": 0.00046982524382767213, |
| "loss": 3.2596, |
| "step": 34700 |
| }, |
| { |
| "epoch": 0.696, |
| "grad_norm": 0.14039067924022675, |
| "learning_rate": 0.0004642290490225486, |
| "loss": 3.2582, |
| "step": 34800 |
| }, |
| { |
| "epoch": 0.698, |
| "grad_norm": 0.14533616602420807, |
| "learning_rate": 0.0004586562905744784, |
| "loss": 3.237, |
| "step": 34900 |
| }, |
| { |
| "epoch": 0.7, |
| "grad_norm": 0.21432434022426605, |
| "learning_rate": 0.00045310721225396854, |
| "loss": 3.2596, |
| "step": 35000 |
| }, |
| { |
| "epoch": 0.7, |
| "eval_accuracy": 0.38947553816046965, |
| "eval_loss": 3.2606313228607178, |
| "eval_runtime": 10.6118, |
| "eval_samples_per_second": 94.235, |
| "eval_steps_per_second": 0.754, |
| "step": 35000 |
| }, |
| { |
| "epoch": 0.702, |
| "grad_norm": 0.15394216775894165, |
| "learning_rate": 0.00044758205679568163, |
| "loss": 3.2522, |
| "step": 35100 |
| }, |
| { |
| "epoch": 0.704, |
| "grad_norm": 0.14649616181850433, |
| "learning_rate": 0.00044208106588781694, |
| "loss": 3.2523, |
| "step": 35200 |
| }, |
| { |
| "epoch": 0.706, |
| "grad_norm": 0.13892528414726257, |
| "learning_rate": 0.00043660448016153656, |
| "loss": 3.2486, |
| "step": 35300 |
| }, |
| { |
| "epoch": 0.708, |
| "grad_norm": 0.16032464802265167, |
| "learning_rate": 0.0004311525391804426, |
| "loss": 3.2464, |
| "step": 35400 |
| }, |
| { |
| "epoch": 0.71, |
| "grad_norm": 0.14114409685134888, |
| "learning_rate": 0.0004257254814300947, |
| "loss": 3.2546, |
| "step": 35500 |
| }, |
| { |
| "epoch": 0.712, |
| "grad_norm": 0.15264350175857544, |
| "learning_rate": 0.000420323544307581, |
| "loss": 3.2436, |
| "step": 35600 |
| }, |
| { |
| "epoch": 0.714, |
| "grad_norm": 0.21544255316257477, |
| "learning_rate": 0.0004149469641111302, |
| "loss": 3.2318, |
| "step": 35700 |
| }, |
| { |
| "epoch": 0.716, |
| "grad_norm": 0.16077803075313568, |
| "learning_rate": 0.00040959597602977806, |
| "loss": 3.2545, |
| "step": 35800 |
| }, |
| { |
| "epoch": 0.718, |
| "grad_norm": 0.14488866925239563, |
| "learning_rate": 0.00040427081413307864, |
| "loss": 3.2526, |
| "step": 35900 |
| }, |
| { |
| "epoch": 0.72, |
| "grad_norm": 0.1400013417005539, |
| "learning_rate": 0.00039897171136086363, |
| "loss": 3.2217, |
| "step": 36000 |
| }, |
| { |
| "epoch": 0.72, |
| "eval_accuracy": 0.3897240704500978, |
| "eval_loss": 3.257254123687744, |
| "eval_runtime": 11.7287, |
| "eval_samples_per_second": 85.261, |
| "eval_steps_per_second": 0.682, |
| "step": 36000 |
| }, |
| { |
| "epoch": 0.722, |
| "grad_norm": 0.14137360453605652, |
| "learning_rate": 0.0003936988995130556, |
| "loss": 3.254, |
| "step": 36100 |
| }, |
| { |
| "epoch": 0.724, |
| "grad_norm": 0.1477884203195572, |
| "learning_rate": 0.0003884526092395261, |
| "loss": 3.25, |
| "step": 36200 |
| }, |
| { |
| "epoch": 0.726, |
| "grad_norm": 0.14579065144062042, |
| "learning_rate": 0.00038323307003000693, |
| "loss": 3.2324, |
| "step": 36300 |
| }, |
| { |
| "epoch": 0.728, |
| "grad_norm": 0.18951904773712158, |
| "learning_rate": 0.0003780405102040524, |
| "loss": 3.2393, |
| "step": 36400 |
| }, |
| { |
| "epoch": 0.73, |
| "grad_norm": 0.1553814709186554, |
| "learning_rate": 0.0003728751569010509, |
| "loss": 3.2329, |
| "step": 36500 |
| }, |
| { |
| "epoch": 0.732, |
| "grad_norm": 0.15148834884166718, |
| "learning_rate": 0.00036773723607028965, |
| "loss": 3.2493, |
| "step": 36600 |
| }, |
| { |
| "epoch": 0.734, |
| "grad_norm": 0.14959374070167542, |
| "learning_rate": 0.00036262697246107, |
| "loss": 3.2412, |
| "step": 36700 |
| }, |
| { |
| "epoch": 0.736, |
| "grad_norm": 0.15129899978637695, |
| "learning_rate": 0.0003575445896128768, |
| "loss": 3.2299, |
| "step": 36800 |
| }, |
| { |
| "epoch": 0.738, |
| "grad_norm": 0.14516687393188477, |
| "learning_rate": 0.0003524903098456013, |
| "loss": 3.2402, |
| "step": 36900 |
| }, |
| { |
| "epoch": 0.74, |
| "grad_norm": 0.14438898861408234, |
| "learning_rate": 0.00034746435424981315, |
| "loss": 3.2538, |
| "step": 37000 |
| }, |
| { |
| "epoch": 0.74, |
| "eval_accuracy": 0.3904481409001957, |
| "eval_loss": 3.250405788421631, |
| "eval_runtime": 11.8201, |
| "eval_samples_per_second": 84.602, |
| "eval_steps_per_second": 0.677, |
| "step": 37000 |
| }, |
| { |
| "epoch": 0.742, |
| "grad_norm": 0.21630696952342987, |
| "learning_rate": 0.00034246694267709257, |
| "loss": 3.2171, |
| "step": 37100 |
| }, |
| { |
| "epoch": 0.744, |
| "grad_norm": 0.14735347032546997, |
| "learning_rate": 0.0003374982937304113, |
| "loss": 3.2346, |
| "step": 37200 |
| }, |
| { |
| "epoch": 0.746, |
| "grad_norm": 0.15532496571540833, |
| "learning_rate": 0.0003325586247545698, |
| "loss": 3.2403, |
| "step": 37300 |
| }, |
| { |
| "epoch": 0.748, |
| "grad_norm": 0.17086553573608398, |
| "learning_rate": 0.00032764815182669205, |
| "loss": 3.2336, |
| "step": 37400 |
| }, |
| { |
| "epoch": 0.75, |
| "grad_norm": 0.14083310961723328, |
| "learning_rate": 0.00032276708974677117, |
| "loss": 3.2287, |
| "step": 37500 |
| }, |
| { |
| "epoch": 0.752, |
| "grad_norm": 0.13698738813400269, |
| "learning_rate": 0.00031791565202827533, |
| "loss": 3.2262, |
| "step": 37600 |
| }, |
| { |
| "epoch": 0.754, |
| "grad_norm": 0.1423436999320984, |
| "learning_rate": 0.0003130940508888063, |
| "loss": 3.2448, |
| "step": 37700 |
| }, |
| { |
| "epoch": 0.756, |
| "grad_norm": 0.2180064171552658, |
| "learning_rate": 0.00030830249724081817, |
| "loss": 3.2235, |
| "step": 37800 |
| }, |
| { |
| "epoch": 0.758, |
| "grad_norm": 0.16625255346298218, |
| "learning_rate": 0.0003035412006823901, |
| "loss": 3.2224, |
| "step": 37900 |
| }, |
| { |
| "epoch": 0.76, |
| "grad_norm": 0.14719747006893158, |
| "learning_rate": 0.0002988103694880576, |
| "loss": 3.2353, |
| "step": 38000 |
| }, |
| { |
| "epoch": 0.76, |
| "eval_accuracy": 0.39081996086105675, |
| "eval_loss": 3.2475223541259766, |
| "eval_runtime": 11.4579, |
| "eval_samples_per_second": 87.276, |
| "eval_steps_per_second": 0.698, |
| "step": 38000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 128, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|