{ "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 215, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.023255813953488372, "grad_norm": 0.42513933777809143, "learning_rate": 2e-05, "loss": 1.4281, "step": 5 }, { "epoch": 0.046511627906976744, "grad_norm": 0.2810695171356201, "learning_rate": 2e-05, "loss": 1.3645, "step": 10 }, { "epoch": 0.06976744186046512, "grad_norm": 0.3605867028236389, "learning_rate": 2e-05, "loss": 1.3975, "step": 15 }, { "epoch": 0.09302325581395349, "grad_norm": 0.26210489869117737, "learning_rate": 2e-05, "loss": 1.2299, "step": 20 }, { "epoch": 0.11627906976744186, "grad_norm": 0.2155284434556961, "learning_rate": 2e-05, "loss": 1.1918, "step": 25 }, { "epoch": 0.13953488372093023, "grad_norm": 0.28256216645240784, "learning_rate": 2e-05, "loss": 1.2193, "step": 30 }, { "epoch": 0.16279069767441862, "grad_norm": 0.24989347159862518, "learning_rate": 2e-05, "loss": 1.2496, "step": 35 }, { "epoch": 0.18604651162790697, "grad_norm": 0.20540378987789154, "learning_rate": 2e-05, "loss": 1.0645, "step": 40 }, { "epoch": 0.20930232558139536, "grad_norm": 0.28998884558677673, "learning_rate": 2e-05, "loss": 1.1089, "step": 45 }, { "epoch": 0.23255813953488372, "grad_norm": 0.3626082241535187, "learning_rate": 2e-05, "loss": 1.2453, "step": 50 }, { "epoch": 0.2558139534883721, "grad_norm": 0.2234840840101242, "learning_rate": 2e-05, "loss": 1.1233, "step": 55 }, { "epoch": 0.27906976744186046, "grad_norm": 0.20808465778827667, "learning_rate": 2e-05, "loss": 1.089, "step": 60 }, { "epoch": 0.3023255813953488, "grad_norm": 0.31903448700904846, "learning_rate": 2e-05, "loss": 1.176, "step": 65 }, { "epoch": 0.32558139534883723, "grad_norm": 0.2401343584060669, "learning_rate": 2e-05, "loss": 1.1098, "step": 70 }, { "epoch": 0.3488372093023256, "grad_norm": 0.2063365876674652, "learning_rate": 2e-05, "loss": 1.105, "step": 75 }, { "epoch": 0.37209302325581395, "grad_norm": 0.3152436912059784, "learning_rate": 2e-05, "loss": 1.0867, "step": 80 }, { "epoch": 0.3953488372093023, "grad_norm": 0.28349247574806213, "learning_rate": 2e-05, "loss": 1.2119, "step": 85 }, { "epoch": 0.4186046511627907, "grad_norm": 0.21781769394874573, "learning_rate": 2e-05, "loss": 1.0341, "step": 90 }, { "epoch": 0.4418604651162791, "grad_norm": 0.2972399890422821, "learning_rate": 2e-05, "loss": 1.0487, "step": 95 }, { "epoch": 0.46511627906976744, "grad_norm": 0.3430672883987427, "learning_rate": 2e-05, "loss": 1.1891, "step": 100 }, { "epoch": 0.4883720930232558, "grad_norm": 0.24118821322917938, "learning_rate": 2e-05, "loss": 1.0056, "step": 105 }, { "epoch": 0.5116279069767442, "grad_norm": 0.23012392222881317, "learning_rate": 2e-05, "loss": 1.0166, "step": 110 }, { "epoch": 0.5348837209302325, "grad_norm": 0.41410136222839355, "learning_rate": 2e-05, "loss": 1.1573, "step": 115 }, { "epoch": 0.5581395348837209, "grad_norm": 0.25534236431121826, "learning_rate": 2e-05, "loss": 1.091, "step": 120 }, { "epoch": 0.5813953488372093, "grad_norm": 0.33041778206825256, "learning_rate": 2e-05, "loss": 1.0053, "step": 125 }, { "epoch": 0.6046511627906976, "grad_norm": 0.3311578333377838, "learning_rate": 2e-05, "loss": 1.0512, "step": 130 }, { "epoch": 0.627906976744186, "grad_norm": 0.3010158836841583, "learning_rate": 2e-05, "loss": 1.1523, "step": 135 }, { "epoch": 0.6511627906976745, "grad_norm": 0.2367880493402481, "learning_rate": 2e-05, "loss": 0.9708, "step": 140 }, { "epoch": 0.6744186046511628, "grad_norm": 0.30969762802124023, "learning_rate": 2e-05, "loss": 1.0445, "step": 145 }, { "epoch": 0.6976744186046512, "grad_norm": 0.43711283802986145, "learning_rate": 2e-05, "loss": 1.1745, "step": 150 }, { "epoch": 0.7209302325581395, "grad_norm": 0.27942749857902527, "learning_rate": 2e-05, "loss": 1.004, "step": 155 }, { "epoch": 0.7441860465116279, "grad_norm": 0.2938617467880249, "learning_rate": 2e-05, "loss": 1.041, "step": 160 }, { "epoch": 0.7674418604651163, "grad_norm": 0.3397826552391052, "learning_rate": 2e-05, "loss": 1.0989, "step": 165 }, { "epoch": 0.7906976744186046, "grad_norm": 0.2756069302558899, "learning_rate": 2e-05, "loss": 1.049, "step": 170 }, { "epoch": 0.813953488372093, "grad_norm": 0.2508189380168915, "learning_rate": 2e-05, "loss": 0.9834, "step": 175 }, { "epoch": 0.8372093023255814, "grad_norm": 0.34858086705207825, "learning_rate": 2e-05, "loss": 1.0219, "step": 180 }, { "epoch": 0.8604651162790697, "grad_norm": 0.30454790592193604, "learning_rate": 2e-05, "loss": 1.1379, "step": 185 }, { "epoch": 0.8837209302325582, "grad_norm": 0.2674161195755005, "learning_rate": 2e-05, "loss": 1.0125, "step": 190 }, { "epoch": 0.9069767441860465, "grad_norm": 0.3410373330116272, "learning_rate": 2e-05, "loss": 0.9154, "step": 195 }, { "epoch": 0.9302325581395349, "grad_norm": 0.3718777298927307, "learning_rate": 2e-05, "loss": 1.1459, "step": 200 }, { "epoch": 0.9534883720930233, "grad_norm": 0.2812870740890503, "learning_rate": 2e-05, "loss": 0.9306, "step": 205 }, { "epoch": 0.9767441860465116, "grad_norm": 0.30063411593437195, "learning_rate": 2e-05, "loss": 0.9894, "step": 210 }, { "epoch": 1.0, "grad_norm": 0.3839293122291565, "learning_rate": 2e-05, "loss": 1.0635, "step": 215 } ], "logging_steps": 5, "max_steps": 215, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 99999, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 3.502029231221637e+17, "train_batch_size": 8, "trial_name": null, "trial_params": null }