{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.23529411764705882, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.023529411764705882, "grad_norm": 0.47367432713508606, "learning_rate": 0.0002, "loss": 0.8419, "step": 50 }, { "epoch": 0.047058823529411764, "grad_norm": 0.4179942309856415, "learning_rate": 0.0002, "loss": 0.554, "step": 100 }, { "epoch": 0.07058823529411765, "grad_norm": 0.40505528450012207, "learning_rate": 0.0002, "loss": 0.5132, "step": 150 }, { "epoch": 0.09411764705882353, "grad_norm": 0.3808290958404541, "learning_rate": 0.0002, "loss": 0.5017, "step": 200 }, { "epoch": 0.11764705882352941, "grad_norm": 0.4120051860809326, "learning_rate": 0.0002, "loss": 0.4731, "step": 250 }, { "epoch": 0.1411764705882353, "grad_norm": 0.3958088755607605, "learning_rate": 0.0002, "loss": 0.4676, "step": 300 }, { "epoch": 0.16470588235294117, "grad_norm": 0.3591330349445343, "learning_rate": 0.0002, "loss": 0.4393, "step": 350 }, { "epoch": 0.18823529411764706, "grad_norm": 0.4021323621273041, "learning_rate": 0.0002, "loss": 0.4328, "step": 400 }, { "epoch": 0.21176470588235294, "grad_norm": 0.3221900165081024, "learning_rate": 0.0002, "loss": 0.4532, "step": 450 }, { "epoch": 0.23529411764705882, "grad_norm": 0.3609311282634735, "learning_rate": 0.0002, "loss": 0.4484, "step": 500 } ], "logging_steps": 50, "max_steps": 2125, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 6.339509608594637e+16, "train_batch_size": 1, "trial_name": null, "trial_params": null }