{ "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 215, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.023255813953488372, "grad_norm": 0.3724094033241272, "learning_rate": 2e-05, "loss": 1.5349, "step": 5 }, { "epoch": 0.046511627906976744, "grad_norm": 0.26067793369293213, "learning_rate": 2e-05, "loss": 1.3942, "step": 10 }, { "epoch": 0.06976744186046512, "grad_norm": 0.3689556419849396, "learning_rate": 2e-05, "loss": 1.3743, "step": 15 }, { "epoch": 0.09302325581395349, "grad_norm": 0.24060697853565216, "learning_rate": 2e-05, "loss": 1.2528, "step": 20 }, { "epoch": 0.11627906976744186, "grad_norm": 0.21089781820774078, "learning_rate": 2e-05, "loss": 1.2474, "step": 25 }, { "epoch": 0.13953488372093023, "grad_norm": 0.27054956555366516, "learning_rate": 2e-05, "loss": 1.2748, "step": 30 }, { "epoch": 0.16279069767441862, "grad_norm": 0.25132614374160767, "learning_rate": 2e-05, "loss": 1.254, "step": 35 }, { "epoch": 0.18604651162790697, "grad_norm": 0.20504344999790192, "learning_rate": 2e-05, "loss": 1.1509, "step": 40 }, { "epoch": 0.20930232558139536, "grad_norm": 0.22182047367095947, "learning_rate": 2e-05, "loss": 1.159, "step": 45 }, { "epoch": 0.23255813953488372, "grad_norm": 0.3401654362678528, "learning_rate": 2e-05, "loss": 1.2327, "step": 50 }, { "epoch": 0.2558139534883721, "grad_norm": 0.22995007038116455, "learning_rate": 2e-05, "loss": 1.0826, "step": 55 }, { "epoch": 0.27906976744186046, "grad_norm": 0.22365306317806244, "learning_rate": 2e-05, "loss": 1.1389, "step": 60 }, { "epoch": 0.3023255813953488, "grad_norm": 0.3217207193374634, "learning_rate": 2e-05, "loss": 1.1764, "step": 65 }, { "epoch": 0.32558139534883723, "grad_norm": 0.28037896752357483, "learning_rate": 2e-05, "loss": 1.1417, "step": 70 }, { "epoch": 0.3488372093023256, "grad_norm": 0.21693165600299835, "learning_rate": 2e-05, "loss": 1.0699, "step": 75 }, { "epoch": 0.37209302325581395, "grad_norm": 0.38298916816711426, "learning_rate": 2e-05, "loss": 1.1218, "step": 80 }, { "epoch": 0.3953488372093023, "grad_norm": 0.2960298955440521, "learning_rate": 2e-05, "loss": 1.1903, "step": 85 }, { "epoch": 0.4186046511627907, "grad_norm": 0.2717413604259491, "learning_rate": 2e-05, "loss": 1.086, "step": 90 }, { "epoch": 0.4418604651162791, "grad_norm": 0.2703586518764496, "learning_rate": 2e-05, "loss": 1.1053, "step": 95 }, { "epoch": 0.46511627906976744, "grad_norm": 0.40586695075035095, "learning_rate": 2e-05, "loss": 1.1559, "step": 100 }, { "epoch": 0.4883720930232558, "grad_norm": 0.24370810389518738, "learning_rate": 2e-05, "loss": 1.088, "step": 105 }, { "epoch": 0.5116279069767442, "grad_norm": 0.23648591339588165, "learning_rate": 2e-05, "loss": 1.0866, "step": 110 }, { "epoch": 0.5348837209302325, "grad_norm": 0.33186522126197815, "learning_rate": 2e-05, "loss": 1.1505, "step": 115 }, { "epoch": 0.5581395348837209, "grad_norm": 0.2656863331794739, "learning_rate": 2e-05, "loss": 1.1016, "step": 120 }, { "epoch": 0.5813953488372093, "grad_norm": 0.2450995147228241, "learning_rate": 2e-05, "loss": 1.0614, "step": 125 }, { "epoch": 0.6046511627906976, "grad_norm": 0.33901068568229675, "learning_rate": 2e-05, "loss": 1.0513, "step": 130 }, { "epoch": 0.627906976744186, "grad_norm": 0.29623690247535706, "learning_rate": 2e-05, "loss": 1.1418, "step": 135 }, { "epoch": 0.6511627906976745, "grad_norm": 0.2740626633167267, "learning_rate": 2e-05, "loss": 1.0751, "step": 140 }, { "epoch": 0.6744186046511628, "grad_norm": 0.3096587657928467, "learning_rate": 2e-05, "loss": 1.1081, "step": 145 }, { "epoch": 0.6976744186046512, "grad_norm": 0.37749382853507996, "learning_rate": 2e-05, "loss": 1.1435, "step": 150 }, { "epoch": 0.7209302325581395, "grad_norm": 0.29796457290649414, "learning_rate": 2e-05, "loss": 1.0263, "step": 155 }, { "epoch": 0.7441860465116279, "grad_norm": 0.2795608341693878, "learning_rate": 2e-05, "loss": 1.0689, "step": 160 }, { "epoch": 0.7674418604651163, "grad_norm": 0.3110540211200714, "learning_rate": 2e-05, "loss": 1.1006, "step": 165 }, { "epoch": 0.7906976744186046, "grad_norm": 0.3028092086315155, "learning_rate": 2e-05, "loss": 1.1031, "step": 170 }, { "epoch": 0.813953488372093, "grad_norm": 0.3238154351711273, "learning_rate": 2e-05, "loss": 1.0344, "step": 175 }, { "epoch": 0.8372093023255814, "grad_norm": 0.30875471234321594, "learning_rate": 2e-05, "loss": 1.0912, "step": 180 }, { "epoch": 0.8604651162790697, "grad_norm": 0.31477639079093933, "learning_rate": 2e-05, "loss": 1.1346, "step": 185 }, { "epoch": 0.8837209302325582, "grad_norm": 0.2805086374282837, "learning_rate": 2e-05, "loss": 1.0635, "step": 190 }, { "epoch": 0.9069767441860465, "grad_norm": 0.34058037400245667, "learning_rate": 2e-05, "loss": 1.0619, "step": 195 }, { "epoch": 0.9302325581395349, "grad_norm": 3.391491651535034, "learning_rate": 2e-05, "loss": 1.154, "step": 200 }, { "epoch": 0.9534883720930233, "grad_norm": 0.31490182876586914, "learning_rate": 2e-05, "loss": 1.0534, "step": 205 }, { "epoch": 0.9767441860465116, "grad_norm": 0.27165260910987854, "learning_rate": 2e-05, "loss": 1.0711, "step": 210 }, { "epoch": 1.0, "grad_norm": 0.3866237699985504, "learning_rate": 2e-05, "loss": 1.1221, "step": 215 } ], "logging_steps": 5, "max_steps": 215, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 99999, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 3.303415539208028e+17, "train_batch_size": 8, "trial_name": null, "trial_params": null }