{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 2.0, "eval_steps": 500, "global_step": 144, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.07032967032967033, "grad_norm": 0.015869140625, "learning_rate": 4.972222222222223e-05, "loss": 0.6071998119354248, "step": 5 }, { "epoch": 0.14065934065934066, "grad_norm": 0.01129150390625, "learning_rate": 4.937500000000001e-05, "loss": 0.6054113388061524, "step": 10 }, { "epoch": 0.210989010989011, "grad_norm": 0.01287841796875, "learning_rate": 4.902777777777778e-05, "loss": 0.6566670417785645, "step": 15 }, { "epoch": 0.2813186813186813, "grad_norm": 0.01220703125, "learning_rate": 4.8680555555555554e-05, "loss": 0.6298727989196777, "step": 20 }, { "epoch": 0.3516483516483517, "grad_norm": 0.013916015625, "learning_rate": 4.8333333333333334e-05, "loss": 0.6060850143432617, "step": 25 }, { "epoch": 0.421978021978022, "grad_norm": 0.01153564453125, "learning_rate": 4.7986111111111113e-05, "loss": 0.5983633995056152, "step": 30 }, { "epoch": 0.49230769230769234, "grad_norm": 0.00970458984375, "learning_rate": 4.7638888888888887e-05, "loss": 0.5858489036560058, "step": 35 }, { "epoch": 0.5626373626373626, "grad_norm": 0.01385498046875, "learning_rate": 4.7291666666666666e-05, "loss": 0.6647251129150391, "step": 40 }, { "epoch": 0.6329670329670329, "grad_norm": 0.01177978515625, "learning_rate": 4.6944444444444446e-05, "loss": 0.606710147857666, "step": 45 }, { "epoch": 0.7032967032967034, "grad_norm": 0.011962890625, "learning_rate": 4.6597222222222226e-05, "loss": 0.5918601512908935, "step": 50 }, { "epoch": 0.7736263736263737, "grad_norm": 0.01129150390625, "learning_rate": 4.6250000000000006e-05, "loss": 0.6076582431793213, "step": 55 }, { "epoch": 0.843956043956044, "grad_norm": 0.00958251953125, "learning_rate": 4.590277777777778e-05, "loss": 0.5553756713867187, "step": 60 }, { "epoch": 0.9142857142857143, "grad_norm": 0.01348876953125, "learning_rate": 4.555555555555556e-05, "loss": 0.6189921379089356, "step": 65 }, { "epoch": 0.9846153846153847, "grad_norm": 0.01220703125, "learning_rate": 4.520833333333334e-05, "loss": 0.5911207675933838, "step": 70 }, { "epoch": 1.0421978021978022, "grad_norm": 0.00958251953125, "learning_rate": 4.486111111111111e-05, "loss": 0.5420764923095703, "step": 75 }, { "epoch": 1.1125274725274725, "grad_norm": 0.01153564453125, "learning_rate": 4.4513888888888885e-05, "loss": 0.4824223518371582, "step": 80 }, { "epoch": 1.1828571428571428, "grad_norm": 0.01434326171875, "learning_rate": 4.4166666666666665e-05, "loss": 0.5018988132476807, "step": 85 }, { "epoch": 1.2531868131868131, "grad_norm": 0.013671875, "learning_rate": 4.3819444444444445e-05, "loss": 0.5336177825927735, "step": 90 }, { "epoch": 1.3235164835164834, "grad_norm": 0.01263427734375, "learning_rate": 4.3472222222222225e-05, "loss": 0.4850775241851807, "step": 95 }, { "epoch": 1.393846153846154, "grad_norm": 0.0126953125, "learning_rate": 4.3125000000000005e-05, "loss": 0.5138346195220947, "step": 100 }, { "epoch": 1.4641758241758243, "grad_norm": 0.01397705078125, "learning_rate": 4.277777777777778e-05, "loss": 0.5070078849792481, "step": 105 }, { "epoch": 1.5345054945054946, "grad_norm": 0.0118408203125, "learning_rate": 4.243055555555556e-05, "loss": 0.5198071479797364, "step": 110 }, { "epoch": 1.6048351648351649, "grad_norm": 0.0137939453125, "learning_rate": 4.208333333333334e-05, "loss": 0.5197066783905029, "step": 115 }, { "epoch": 1.6751648351648352, "grad_norm": 0.00970458984375, "learning_rate": 4.173611111111112e-05, "loss": 0.4952798366546631, "step": 120 }, { "epoch": 1.7454945054945055, "grad_norm": 0.01190185546875, "learning_rate": 4.138888888888889e-05, "loss": 0.5169046878814697, "step": 125 }, { "epoch": 1.8158241758241758, "grad_norm": 0.01324462890625, "learning_rate": 4.104166666666667e-05, "loss": 0.5308417320251465, "step": 130 }, { "epoch": 1.8861538461538463, "grad_norm": 0.0146484375, "learning_rate": 4.0694444444444444e-05, "loss": 0.46563105583190917, "step": 135 }, { "epoch": 1.9564835164835164, "grad_norm": 0.020751953125, "learning_rate": 4.0347222222222223e-05, "loss": 0.521865701675415, "step": 140 } ], "logging_steps": 5, "max_steps": 720, "num_input_tokens_seen": 0, "num_train_epochs": 10, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 3.694778508115968e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }