{ "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 215, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.023255813953488372, "grad_norm": 1.7071588039398193, "learning_rate": 2e-05, "loss": 1.2071, "step": 5 }, { "epoch": 0.046511627906976744, "grad_norm": 0.3188117742538452, "learning_rate": 2e-05, "loss": 1.0864, "step": 10 }, { "epoch": 0.06976744186046512, "grad_norm": 0.2328031063079834, "learning_rate": 2e-05, "loss": 0.9077, "step": 15 }, { "epoch": 0.09302325581395349, "grad_norm": 0.3528258800506592, "learning_rate": 2e-05, "loss": 0.951, "step": 20 }, { "epoch": 0.11627906976744186, "grad_norm": 0.168439120054245, "learning_rate": 2e-05, "loss": 0.9764, "step": 25 }, { "epoch": 0.13953488372093023, "grad_norm": 0.14444082975387573, "learning_rate": 2e-05, "loss": 0.8102, "step": 30 }, { "epoch": 0.16279069767441862, "grad_norm": 0.2678701877593994, "learning_rate": 2e-05, "loss": 0.8644, "step": 35 }, { "epoch": 0.18604651162790697, "grad_norm": 0.2088869959115982, "learning_rate": 2e-05, "loss": 0.9437, "step": 40 }, { "epoch": 0.20930232558139536, "grad_norm": 0.14392392337322235, "learning_rate": 2e-05, "loss": 0.8126, "step": 45 }, { "epoch": 0.23255813953488372, "grad_norm": 0.31939807534217834, "learning_rate": 2e-05, "loss": 0.747, "step": 50 }, { "epoch": 0.2558139534883721, "grad_norm": 0.21644248068332672, "learning_rate": 2e-05, "loss": 0.908, "step": 55 }, { "epoch": 0.27906976744186046, "grad_norm": 0.14742884039878845, "learning_rate": 2e-05, "loss": 0.8698, "step": 60 }, { "epoch": 0.3023255813953488, "grad_norm": 0.18430279195308685, "learning_rate": 2e-05, "loss": 0.7584, "step": 65 }, { "epoch": 0.32558139534883723, "grad_norm": 0.3597599267959595, "learning_rate": 2e-05, "loss": 0.8581, "step": 70 }, { "epoch": 0.3488372093023256, "grad_norm": 0.15492893755435944, "learning_rate": 2e-05, "loss": 0.9204, "step": 75 }, { "epoch": 0.37209302325581395, "grad_norm": 0.147719144821167, "learning_rate": 2e-05, "loss": 0.7454, "step": 80 }, { "epoch": 0.3953488372093023, "grad_norm": 0.246868297457695, "learning_rate": 2e-05, "loss": 0.8537, "step": 85 }, { "epoch": 0.4186046511627907, "grad_norm": 0.17558737099170685, "learning_rate": 2e-05, "loss": 0.9411, "step": 90 }, { "epoch": 0.4418604651162791, "grad_norm": 0.15025165677070618, "learning_rate": 2e-05, "loss": 0.7557, "step": 95 }, { "epoch": 0.46511627906976744, "grad_norm": 0.29634204506874084, "learning_rate": 2e-05, "loss": 0.7418, "step": 100 }, { "epoch": 0.4883720930232558, "grad_norm": 0.18677298724651337, "learning_rate": 2e-05, "loss": 0.9445, "step": 105 }, { "epoch": 0.5116279069767442, "grad_norm": 0.1614491045475006, "learning_rate": 2e-05, "loss": 0.8228, "step": 110 }, { "epoch": 0.5348837209302325, "grad_norm": 0.18401965498924255, "learning_rate": 2e-05, "loss": 0.7213, "step": 115 }, { "epoch": 0.5581395348837209, "grad_norm": 0.22585374116897583, "learning_rate": 2e-05, "loss": 0.8905, "step": 120 }, { "epoch": 0.5813953488372093, "grad_norm": 0.1724023073911667, "learning_rate": 2e-05, "loss": 0.8981, "step": 125 }, { "epoch": 0.6046511627906976, "grad_norm": 0.14925441145896912, "learning_rate": 2e-05, "loss": 0.7038, "step": 130 }, { "epoch": 0.627906976744186, "grad_norm": 0.2597886919975281, "learning_rate": 2e-05, "loss": 0.7908, "step": 135 }, { "epoch": 0.6511627906976745, "grad_norm": 0.17634142935276031, "learning_rate": 2e-05, "loss": 0.9247, "step": 140 }, { "epoch": 0.6744186046511628, "grad_norm": 0.16650615632534027, "learning_rate": 2e-05, "loss": 0.7638, "step": 145 }, { "epoch": 0.6976744186046512, "grad_norm": 0.2985098361968994, "learning_rate": 2e-05, "loss": 0.7397, "step": 150 }, { "epoch": 0.7209302325581395, "grad_norm": 0.20995645225048065, "learning_rate": 2e-05, "loss": 0.9161, "step": 155 }, { "epoch": 0.7441860465116279, "grad_norm": 0.16316258907318115, "learning_rate": 2e-05, "loss": 0.8105, "step": 160 }, { "epoch": 0.7674418604651163, "grad_norm": 0.18892395496368408, "learning_rate": 2e-05, "loss": 0.6916, "step": 165 }, { "epoch": 0.7906976744186046, "grad_norm": 0.23752492666244507, "learning_rate": 2e-05, "loss": 0.8452, "step": 170 }, { "epoch": 0.813953488372093, "grad_norm": 0.18073770403862, "learning_rate": 2e-05, "loss": 0.8764, "step": 175 }, { "epoch": 0.8372093023255814, "grad_norm": 0.15237540006637573, "learning_rate": 2e-05, "loss": 0.7132, "step": 180 }, { "epoch": 0.8604651162790697, "grad_norm": 0.294465571641922, "learning_rate": 2e-05, "loss": 0.8096, "step": 185 }, { "epoch": 0.8837209302325582, "grad_norm": 0.263343870639801, "learning_rate": 2e-05, "loss": 0.8742, "step": 190 }, { "epoch": 0.9069767441860465, "grad_norm": 0.16255247592926025, "learning_rate": 2e-05, "loss": 0.733, "step": 195 }, { "epoch": 0.9302325581395349, "grad_norm": 0.33820098638534546, "learning_rate": 2e-05, "loss": 0.7324, "step": 200 }, { "epoch": 0.9534883720930233, "grad_norm": 0.2312634140253067, "learning_rate": 2e-05, "loss": 0.9044, "step": 205 }, { "epoch": 0.9767441860465116, "grad_norm": 0.2246032953262329, "learning_rate": 2e-05, "loss": 0.8323, "step": 210 }, { "epoch": 1.0, "grad_norm": 0.20284949243068695, "learning_rate": 2e-05, "loss": 0.6975, "step": 215 } ], "logging_steps": 5, "max_steps": 215, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 99999, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 2.8126845911380787e+17, "train_batch_size": 8, "trial_name": null, "trial_params": null }