{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.9927536231884058, "eval_steps": 20, "global_step": 137, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.036231884057971016, "grad_norm": 1.5703125, "learning_rate": 3.139372514179859e-05, "loss": 2.648, "step": 5 }, { "epoch": 0.07246376811594203, "grad_norm": 1.6015625, "learning_rate": 7.063588156904684e-05, "loss": 2.5698, "step": 10 }, { "epoch": 0.10869565217391304, "grad_norm": 1.328125, "learning_rate": 9.417686158405852e-05, "loss": 2.535, "step": 15 }, { "epoch": 0.14492753623188406, "grad_norm": 1.109375, "learning_rate": 9.412834297056845e-05, "loss": 2.5239, "step": 20 }, { "epoch": 0.18115942028985507, "grad_norm": 1.125, "learning_rate": 9.402598775863905e-05, "loss": 2.5369, "step": 25 }, { "epoch": 0.21739130434782608, "grad_norm": 1.125, "learning_rate": 9.386995220630276e-05, "loss": 2.5074, "step": 30 }, { "epoch": 0.2536231884057971, "grad_norm": 1.1015625, "learning_rate": 9.366047452134543e-05, "loss": 2.4874, "step": 35 }, { "epoch": 0.2898550724637681, "grad_norm": 1.1328125, "learning_rate": 9.339787449765238e-05, "loss": 2.5084, "step": 40 }, { "epoch": 0.32608695652173914, "grad_norm": 1.0234375, "learning_rate": 9.308255302700287e-05, "loss": 2.4878, "step": 45 }, { "epoch": 0.36231884057971014, "grad_norm": 1.046875, "learning_rate": 9.271499148705883e-05, "loss": 2.4854, "step": 50 }, { "epoch": 0.39855072463768115, "grad_norm": 1.0703125, "learning_rate": 9.229575100648155e-05, "loss": 2.501, "step": 55 }, { "epoch": 0.43478260869565216, "grad_norm": 1.03125, "learning_rate": 9.182547160829867e-05, "loss": 2.461, "step": 60 }, { "epoch": 0.47101449275362317, "grad_norm": 1.0, "learning_rate": 9.130487123282887e-05, "loss": 2.4792, "step": 65 }, { "epoch": 0.5072463768115942, "grad_norm": 0.9921875, "learning_rate": 9.073474464165638e-05, "loss": 2.4696, "step": 70 }, { "epoch": 0.5434782608695652, "grad_norm": 0.98046875, "learning_rate": 9.011596220432776e-05, "loss": 2.4758, "step": 75 }, { "epoch": 0.5797101449275363, "grad_norm": 0.953125, "learning_rate": 8.944946856962415e-05, "loss": 2.4634, "step": 80 }, { "epoch": 0.6159420289855072, "grad_norm": 0.875, "learning_rate": 8.873628122343667e-05, "loss": 2.4623, "step": 85 }, { "epoch": 0.6521739130434783, "grad_norm": 0.94140625, "learning_rate": 8.797748893544693e-05, "loss": 2.4326, "step": 90 }, { "epoch": 0.6884057971014492, "grad_norm": 0.93359375, "learning_rate": 8.717425009698402e-05, "loss": 2.4757, "step": 95 }, { "epoch": 0.7246376811594203, "grad_norm": 0.953125, "learning_rate": 8.632779095259514e-05, "loss": 2.4531, "step": 100 }, { "epoch": 0.7608695652173914, "grad_norm": 0.90625, "learning_rate": 8.54394037280301e-05, "loss": 2.4681, "step": 105 }, { "epoch": 0.7971014492753623, "grad_norm": 0.921875, "learning_rate": 8.45104446574969e-05, "loss": 2.4615, "step": 110 }, { "epoch": 0.8333333333333334, "grad_norm": 0.93359375, "learning_rate": 8.35423319132008e-05, "loss": 2.4599, "step": 115 }, { "epoch": 0.8695652173913043, "grad_norm": 1.0390625, "learning_rate": 8.253654344032692e-05, "loss": 2.4509, "step": 120 }, { "epoch": 0.8695652173913043, "eval_loss": 2.4220075607299805, "eval_runtime": 13.5232, "eval_samples_per_second": 14.715, "eval_steps_per_second": 14.715, "step": 120 }, { "epoch": 0.9057971014492754, "grad_norm": 1.015625, "learning_rate": 8.149461470077207e-05, "loss": 2.449, "step": 125 }, { "epoch": 0.9420289855072463, "grad_norm": 0.84375, "learning_rate": 8.041813632907031e-05, "loss": 2.4368, "step": 130 }, { "epoch": 0.9782608695652174, "grad_norm": 0.91015625, "learning_rate": 7.930875170409012e-05, "loss": 2.4394, "step": 135 }, { "epoch": 0.9927536231884058, "eval_loss": 2.4152843952178955, "eval_runtime": 12.1031, "eval_samples_per_second": 16.442, "eval_steps_per_second": 16.442, "step": 137 } ], "logging_steps": 5, "max_steps": 414, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 20, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.792294366347264e+17, "train_batch_size": 100, "trial_name": null, "trial_params": null }