{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.5, "eval_steps": 50, "global_step": 100, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.025, "grad_norm": 0.10900866985321045, "learning_rate": 1.6e-06, "loss": 11.06564712524414, "step": 5 }, { "epoch": 0.05, "grad_norm": 0.10152705758810043, "learning_rate": 3.6e-06, "loss": 11.040735626220703, "step": 10 }, { "epoch": 0.075, "grad_norm": 0.09875829517841339, "learning_rate": 3.995627254437549e-06, "loss": 11.026320648193359, "step": 15 }, { "epoch": 0.1, "grad_norm": 0.09957622736692429, "learning_rate": 3.9778957412029366e-06, "loss": 11.012971496582031, "step": 20 }, { "epoch": 0.125, "grad_norm": 0.1074257642030716, "learning_rate": 3.9466531944960116e-06, "loss": 10.992045593261718, "step": 25 }, { "epoch": 0.15, "grad_norm": 0.110373854637146, "learning_rate": 3.902113032590307e-06, "loss": 10.97699432373047, "step": 30 }, { "epoch": 0.175, "grad_norm": 0.11517436057329178, "learning_rate": 3.844579509954769e-06, "loss": 10.947827911376953, "step": 35 }, { "epoch": 0.2, "grad_norm": 0.1078442707657814, "learning_rate": 3.7744456388872234e-06, "loss": 10.949494171142579, "step": 40 }, { "epoch": 0.225, "grad_norm": 0.1252269148826599, "learning_rate": 3.6921905048418706e-06, "loss": 10.916586303710938, "step": 45 }, { "epoch": 0.25, "grad_norm": 0.12218247354030609, "learning_rate": 3.598375993789849e-06, "loss": 10.902359008789062, "step": 50 }, { "epoch": 0.25, "eval_accuracy": 0.001978021978021978, "eval_loss": 10.892147064208984, "eval_runtime": 11.0414, "eval_samples_per_second": 9.057, "eval_steps_per_second": 1.54, "step": 50 }, { "epoch": 0.275, "grad_norm": 0.12842494249343872, "learning_rate": 3.4936429539683075e-06, "loss": 10.880451202392578, "step": 55 }, { "epoch": 0.3, "grad_norm": 0.12605500221252441, "learning_rate": 3.3787068182371314e-06, "loss": 10.865975952148437, "step": 60 }, { "epoch": 0.325, "grad_norm": 0.12930333614349365, "learning_rate": 3.254352716947074e-06, "loss": 10.846955871582031, "step": 65 }, { "epoch": 0.35, "grad_norm": 0.1300428807735443, "learning_rate": 3.1214301147033453e-06, "loss": 10.831236267089844, "step": 70 }, { "epoch": 0.375, "grad_norm": 0.13403981924057007, "learning_rate": 2.9808470076610163e-06, "loss": 10.815153503417969, "step": 75 }, { "epoch": 0.4, "grad_norm": 0.1414586752653122, "learning_rate": 2.833563720990581e-06, "loss": 10.798695373535157, "step": 80 }, { "epoch": 0.425, "grad_norm": 0.1405540555715561, "learning_rate": 2.680586348883286e-06, "loss": 10.79158706665039, "step": 85 }, { "epoch": 0.45, "grad_norm": 0.15637332201004028, "learning_rate": 2.5229598819076393e-06, "loss": 10.773637390136718, "step": 90 }, { "epoch": 0.475, "grad_norm": 0.15481412410736084, "learning_rate": 2.36176106866422e-06, "loss": 10.747270965576172, "step": 95 }, { "epoch": 0.5, "grad_norm": 0.1801270991563797, "learning_rate": 2.198091060500926e-06, "loss": 10.725272369384765, "step": 100 }, { "epoch": 0.5, "eval_accuracy": 0.032405372405372404, "eval_loss": 10.71803092956543, "eval_runtime": 10.5701, "eval_samples_per_second": 9.461, "eval_steps_per_second": 1.608, "step": 100 } ], "logging_steps": 5, "max_steps": 200, "num_input_tokens_seen": 0, "num_train_epochs": 9223372036854775807, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 6, "trial_name": null, "trial_params": null }