{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.14566783848350068, "eval_steps": 500, "global_step": 15000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.004855594616116689, "grad_norm": 3.0736300945281982, "learning_rate": 4.991907342306472e-05, "loss": 5.2731, "step": 500 }, { "epoch": 0.009711189232233379, "grad_norm": 10.307616233825684, "learning_rate": 4.983814684612945e-05, "loss": 5.2117, "step": 1000 }, { "epoch": 0.014566783848350069, "grad_norm": 15.462577819824219, "learning_rate": 4.9757220269194165e-05, "loss": 5.1878, "step": 1500 }, { "epoch": 0.019422378464466757, "grad_norm": 13.30544376373291, "learning_rate": 4.967629369225889e-05, "loss": 5.1723, "step": 2000 }, { "epoch": 0.024277973080583447, "grad_norm": 13.564447402954102, "learning_rate": 4.959536711532361e-05, "loss": 5.1663, "step": 2500 }, { "epoch": 0.029133567696700138, "grad_norm": 15.116345405578613, "learning_rate": 4.9514440538388335e-05, "loss": 5.1597, "step": 3000 }, { "epoch": 0.033989162312816824, "grad_norm": 9.680214881896973, "learning_rate": 4.9433513961453054e-05, "loss": 5.1509, "step": 3500 }, { "epoch": 0.038844756928933515, "grad_norm": 12.354632377624512, "learning_rate": 4.935258738451778e-05, "loss": 5.1455, "step": 4000 }, { "epoch": 0.043700351545050205, "grad_norm": 14.90212345123291, "learning_rate": 4.92716608075825e-05, "loss": 5.1417, "step": 4500 }, { "epoch": 0.048555946161166895, "grad_norm": 12.882039070129395, "learning_rate": 4.9190734230647223e-05, "loss": 5.1371, "step": 5000 }, { "epoch": 0.053411540777283585, "grad_norm": 15.77685546875, "learning_rate": 4.910980765371194e-05, "loss": 5.1317, "step": 5500 }, { "epoch": 0.058267135393400275, "grad_norm": 11.873464584350586, "learning_rate": 4.902888107677667e-05, "loss": 5.1265, "step": 6000 }, { "epoch": 0.06312273000951696, "grad_norm": 14.090598106384277, "learning_rate": 4.8947954499841386e-05, "loss": 5.1176, "step": 6500 }, { "epoch": 0.06797832462563365, "grad_norm": 13.959632873535156, "learning_rate": 4.8867027922906105e-05, "loss": 5.1135, "step": 7000 }, { "epoch": 0.07283391924175034, "grad_norm": 13.812936782836914, "learning_rate": 4.878610134597083e-05, "loss": 5.1067, "step": 7500 }, { "epoch": 0.07768951385786703, "grad_norm": 10.268146514892578, "learning_rate": 4.870517476903555e-05, "loss": 5.1056, "step": 8000 }, { "epoch": 0.08254510847398372, "grad_norm": 13.129918098449707, "learning_rate": 4.8624248192100275e-05, "loss": 5.0994, "step": 8500 }, { "epoch": 0.08740070309010041, "grad_norm": 15.821222305297852, "learning_rate": 4.8543321615164994e-05, "loss": 5.0889, "step": 9000 }, { "epoch": 0.0922562977062171, "grad_norm": 7.900857448577881, "learning_rate": 4.846239503822972e-05, "loss": 5.0827, "step": 9500 }, { "epoch": 0.09711189232233379, "grad_norm": 5.5381903648376465, "learning_rate": 4.838146846129444e-05, "loss": 5.1042, "step": 10000 }, { "epoch": 0.10196748693845048, "grad_norm": 4.201797008514404, "learning_rate": 4.830054188435916e-05, "loss": 5.0793, "step": 10500 }, { "epoch": 0.10682308155456717, "grad_norm": 8.572744369506836, "learning_rate": 4.821961530742388e-05, "loss": 5.0727, "step": 11000 }, { "epoch": 0.11167867617068386, "grad_norm": 10.609322547912598, "learning_rate": 4.813868873048861e-05, "loss": 5.0683, "step": 11500 }, { "epoch": 0.11653427078680055, "grad_norm": 9.953619003295898, "learning_rate": 4.8057762153553326e-05, "loss": 5.0632, "step": 12000 }, { "epoch": 0.12138986540291724, "grad_norm": 8.941683769226074, "learning_rate": 4.7976835576618045e-05, "loss": 5.0614, "step": 12500 }, { "epoch": 0.12624546001903392, "grad_norm": 9.56987190246582, "learning_rate": 4.789590899968277e-05, "loss": 5.0637, "step": 13000 }, { "epoch": 0.13110105463515062, "grad_norm": 7.185657978057861, "learning_rate": 4.781498242274749e-05, "loss": 5.0569, "step": 13500 }, { "epoch": 0.1359566492512673, "grad_norm": 16.51009178161621, "learning_rate": 4.7734055845812215e-05, "loss": 5.0581, "step": 14000 }, { "epoch": 0.140812243867384, "grad_norm": 10.030774116516113, "learning_rate": 4.7653129268876933e-05, "loss": 5.0552, "step": 14500 }, { "epoch": 0.14566783848350068, "grad_norm": 7.811285018920898, "learning_rate": 4.757220269194166e-05, "loss": 5.0521, "step": 15000 } ], "logging_steps": 500, "max_steps": 308922, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 15000, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 8.520669651145851e+17, "train_batch_size": 2048, "trial_name": null, "trial_params": null }