{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.36156149370092083, "eval_steps": 500, "global_step": 400, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.009039037342523022, "grad_norm": 7.855499744415283, "learning_rate": 6.000000000000001e-07, "loss": 1.4181, "step": 10 }, { "epoch": 0.018078074685046044, "grad_norm": 3.3778610229492188, "learning_rate": 1.2666666666666669e-06, "loss": 1.2494, "step": 20 }, { "epoch": 0.027117112027569064, "grad_norm": 2.6225311756134033, "learning_rate": 1.9333333333333336e-06, "loss": 1.0602, "step": 30 }, { "epoch": 0.03615614937009209, "grad_norm": 1.594939112663269, "learning_rate": 2.6e-06, "loss": 0.9123, "step": 40 }, { "epoch": 0.045195186712615104, "grad_norm": 0.9766483902931213, "learning_rate": 3.266666666666667e-06, "loss": 0.7841, "step": 50 }, { "epoch": 0.05423422405513813, "grad_norm": 0.43368658423423767, "learning_rate": 3.9333333333333335e-06, "loss": 0.6803, "step": 60 }, { "epoch": 0.06327326139766115, "grad_norm": 0.49014410376548767, "learning_rate": 4.600000000000001e-06, "loss": 0.5921, "step": 70 }, { "epoch": 0.07231229874018417, "grad_norm": 0.46368488669395447, "learning_rate": 5.2666666666666665e-06, "loss": 0.6214, "step": 80 }, { "epoch": 0.0813513360827072, "grad_norm": 0.46477940678596497, "learning_rate": 5.933333333333335e-06, "loss": 0.6157, "step": 90 }, { "epoch": 0.09039037342523021, "grad_norm": 0.4853752553462982, "learning_rate": 6.600000000000001e-06, "loss": 0.6628, "step": 100 }, { "epoch": 0.09942941076775323, "grad_norm": 0.36940082907676697, "learning_rate": 7.266666666666668e-06, "loss": 0.5775, "step": 110 }, { "epoch": 0.10846844811027626, "grad_norm": 0.34785982966423035, "learning_rate": 7.933333333333334e-06, "loss": 0.6343, "step": 120 }, { "epoch": 0.11750748545279928, "grad_norm": 0.4787227511405945, "learning_rate": 8.6e-06, "loss": 0.6016, "step": 130 }, { "epoch": 0.1265465227953223, "grad_norm": 0.4643513262271881, "learning_rate": 9.266666666666667e-06, "loss": 0.6388, "step": 140 }, { "epoch": 0.1355855601378453, "grad_norm": 0.384676456451416, "learning_rate": 9.933333333333334e-06, "loss": 0.5895, "step": 150 }, { "epoch": 0.14462459748036835, "grad_norm": 0.4324859380722046, "learning_rate": 9.99991503499922e-06, "loss": 0.5942, "step": 160 }, { "epoch": 0.15366363482289136, "grad_norm": 0.41830965876579285, "learning_rate": 9.99962133253095e-06, "loss": 0.6121, "step": 170 }, { "epoch": 0.1627026721654144, "grad_norm": 0.3795354962348938, "learning_rate": 9.999117855964797e-06, "loss": 0.6041, "step": 180 }, { "epoch": 0.1717417095079374, "grad_norm": 0.28756389021873474, "learning_rate": 9.998404626425627e-06, "loss": 0.5529, "step": 190 }, { "epoch": 0.18078074685046042, "grad_norm": 0.40732458233833313, "learning_rate": 9.997481673839125e-06, "loss": 0.6321, "step": 200 }, { "epoch": 0.18981978419298345, "grad_norm": 0.3699369430541992, "learning_rate": 9.996349036930533e-06, "loss": 0.5886, "step": 210 }, { "epoch": 0.19885882153550646, "grad_norm": 0.4334062337875366, "learning_rate": 9.995006763223028e-06, "loss": 0.646, "step": 220 }, { "epoch": 0.2078978588780295, "grad_norm": 0.3638211786746979, "learning_rate": 9.993454909035724e-06, "loss": 0.5996, "step": 230 }, { "epoch": 0.2169368962205525, "grad_norm": 0.377408891916275, "learning_rate": 9.991693539481317e-06, "loss": 0.5999, "step": 240 }, { "epoch": 0.22597593356307552, "grad_norm": 0.4337885081768036, "learning_rate": 9.989722728463345e-06, "loss": 0.6233, "step": 250 }, { "epoch": 0.23501497090559856, "grad_norm": 0.35624244809150696, "learning_rate": 9.98754255867309e-06, "loss": 0.6279, "step": 260 }, { "epoch": 0.24405400824812157, "grad_norm": 0.385580837726593, "learning_rate": 9.985153121586111e-06, "loss": 0.5896, "step": 270 }, { "epoch": 0.2530930455906446, "grad_norm": 0.36690962314605713, "learning_rate": 9.982554517458403e-06, "loss": 0.6212, "step": 280 }, { "epoch": 0.26213208293316764, "grad_norm": 0.3190907835960388, "learning_rate": 9.979746855322192e-06, "loss": 0.6019, "step": 290 }, { "epoch": 0.2711711202756906, "grad_norm": 0.33014318346977234, "learning_rate": 9.976730252981354e-06, "loss": 0.5906, "step": 300 }, { "epoch": 0.28021015761821366, "grad_norm": 0.4516271650791168, "learning_rate": 9.973504837006487e-06, "loss": 0.612, "step": 310 }, { "epoch": 0.2892491949607367, "grad_norm": 0.3498607873916626, "learning_rate": 9.97007074272958e-06, "loss": 0.5591, "step": 320 }, { "epoch": 0.2982882323032597, "grad_norm": 0.4618026316165924, "learning_rate": 9.966428114238353e-06, "loss": 0.5808, "step": 330 }, { "epoch": 0.3073272696457827, "grad_norm": 0.35378628969192505, "learning_rate": 9.962577104370206e-06, "loss": 0.5983, "step": 340 }, { "epoch": 0.31636630698830576, "grad_norm": 0.3605310916900635, "learning_rate": 9.958517874705793e-06, "loss": 0.5898, "step": 350 }, { "epoch": 0.3254053443308288, "grad_norm": 0.41437646746635437, "learning_rate": 9.954250595562267e-06, "loss": 0.5746, "step": 360 }, { "epoch": 0.3344443816733518, "grad_norm": 0.3715222477912903, "learning_rate": 9.949775445986112e-06, "loss": 0.5994, "step": 370 }, { "epoch": 0.3434834190158748, "grad_norm": 0.41986992955207825, "learning_rate": 9.945092613745642e-06, "loss": 0.5492, "step": 380 }, { "epoch": 0.35252245635839785, "grad_norm": 0.3996647000312805, "learning_rate": 9.940202295323116e-06, "loss": 0.6066, "step": 390 }, { "epoch": 0.36156149370092083, "grad_norm": 0.3808385729789734, "learning_rate": 9.935104695906506e-06, "loss": 0.5628, "step": 400 } ], "logging_steps": 10, "max_steps": 5000, "num_input_tokens_seen": 0, "num_train_epochs": 5, "save_steps": 200, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.2673534264901837e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }