{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 500, "global_step": 204, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.07428040854224698, "grad_norm": 0.0166015625, "learning_rate": 4.970588235294118e-05, "loss": 0.6377665996551514, "step": 5 }, { "epoch": 0.14856081708449395, "grad_norm": 0.0146484375, "learning_rate": 4.933823529411765e-05, "loss": 0.6021791934967041, "step": 10 }, { "epoch": 0.22284122562674094, "grad_norm": 0.01373291015625, "learning_rate": 4.897058823529412e-05, "loss": 0.6274982929229737, "step": 15 }, { "epoch": 0.2971216341689879, "grad_norm": 0.0135498046875, "learning_rate": 4.860294117647059e-05, "loss": 0.599825382232666, "step": 20 }, { "epoch": 0.3714020427112349, "grad_norm": 0.0159912109375, "learning_rate": 4.823529411764706e-05, "loss": 0.6859479427337647, "step": 25 }, { "epoch": 0.4456824512534819, "grad_norm": 0.01422119140625, "learning_rate": 4.7867647058823535e-05, "loss": 0.636006212234497, "step": 30 }, { "epoch": 0.5199628597957289, "grad_norm": 0.01116943359375, "learning_rate": 4.75e-05, "loss": 0.6127719879150391, "step": 35 }, { "epoch": 0.5942432683379758, "grad_norm": 0.01531982421875, "learning_rate": 4.713235294117647e-05, "loss": 0.5927554130554199, "step": 40 }, { "epoch": 0.6685236768802229, "grad_norm": 0.011474609375, "learning_rate": 4.6764705882352944e-05, "loss": 0.6282303333282471, "step": 45 }, { "epoch": 0.7428040854224698, "grad_norm": 0.015869140625, "learning_rate": 4.639705882352942e-05, "loss": 0.5779662609100342, "step": 50 }, { "epoch": 0.8170844939647168, "grad_norm": 0.01318359375, "learning_rate": 4.6029411764705885e-05, "loss": 0.617146348953247, "step": 55 }, { "epoch": 0.8913649025069638, "grad_norm": 0.0130615234375, "learning_rate": 4.566176470588235e-05, "loss": 0.5746903896331788, "step": 60 }, { "epoch": 0.9656453110492108, "grad_norm": 0.0108642578125, "learning_rate": 4.5294117647058826e-05, "loss": 0.5909358024597168, "step": 65 }, { "epoch": 1.0297121634168989, "grad_norm": 0.00994873046875, "learning_rate": 4.49264705882353e-05, "loss": 0.5567221164703369, "step": 70 }, { "epoch": 1.1039925719591457, "grad_norm": 0.01031494140625, "learning_rate": 4.455882352941177e-05, "loss": 0.5583184242248536, "step": 75 }, { "epoch": 1.1782729805013927, "grad_norm": 0.01214599609375, "learning_rate": 4.4191176470588235e-05, "loss": 0.5047792911529541, "step": 80 }, { "epoch": 1.2525533890436398, "grad_norm": 0.014892578125, "learning_rate": 4.382352941176471e-05, "loss": 0.509721040725708, "step": 85 }, { "epoch": 1.3268337975858868, "grad_norm": 0.0162353515625, "learning_rate": 4.345588235294118e-05, "loss": 0.522771167755127, "step": 90 }, { "epoch": 1.4011142061281336, "grad_norm": 0.01263427734375, "learning_rate": 4.308823529411765e-05, "loss": 0.48066258430480957, "step": 95 }, { "epoch": 1.4753946146703807, "grad_norm": 0.02294921875, "learning_rate": 4.272058823529412e-05, "loss": 0.5085556030273437, "step": 100 }, { "epoch": 1.5496750232126275, "grad_norm": 0.01287841796875, "learning_rate": 4.235294117647059e-05, "loss": 0.4806520462036133, "step": 105 }, { "epoch": 1.6239554317548746, "grad_norm": 0.0107421875, "learning_rate": 4.198529411764706e-05, "loss": 0.48960013389587403, "step": 110 }, { "epoch": 1.6982358402971216, "grad_norm": 0.0120849609375, "learning_rate": 4.161764705882353e-05, "loss": 0.47647743225097655, "step": 115 }, { "epoch": 1.7725162488393686, "grad_norm": 0.0230712890625, "learning_rate": 4.125e-05, "loss": 0.5240037918090821, "step": 120 }, { "epoch": 1.8467966573816157, "grad_norm": 0.0152587890625, "learning_rate": 4.0882352941176474e-05, "loss": 0.5106610298156739, "step": 125 }, { "epoch": 1.9210770659238627, "grad_norm": 0.01312255859375, "learning_rate": 4.051470588235294e-05, "loss": 0.5423622608184815, "step": 130 }, { "epoch": 1.9953574744661096, "grad_norm": 0.01214599609375, "learning_rate": 4.0147058823529415e-05, "loss": 0.48050861358642577, "step": 135 }, { "epoch": 2.0594243268337977, "grad_norm": 0.01177978515625, "learning_rate": 3.977941176470588e-05, "loss": 0.4528938293457031, "step": 140 }, { "epoch": 2.1337047353760448, "grad_norm": 0.0135498046875, "learning_rate": 3.9411764705882356e-05, "loss": 0.4708698749542236, "step": 145 }, { "epoch": 2.2079851439182914, "grad_norm": 0.015869140625, "learning_rate": 3.9044117647058823e-05, "loss": 0.40610790252685547, "step": 150 }, { "epoch": 2.2822655524605384, "grad_norm": 0.020263671875, "learning_rate": 3.86764705882353e-05, "loss": 0.4004369735717773, "step": 155 }, { "epoch": 2.3565459610027855, "grad_norm": 0.01495361328125, "learning_rate": 3.830882352941177e-05, "loss": 0.402101993560791, "step": 160 }, { "epoch": 2.4308263695450325, "grad_norm": 0.01519775390625, "learning_rate": 3.794117647058824e-05, "loss": 0.41231889724731446, "step": 165 }, { "epoch": 2.5051067780872796, "grad_norm": 0.01611328125, "learning_rate": 3.7573529411764706e-05, "loss": 0.4784365653991699, "step": 170 }, { "epoch": 2.5793871866295266, "grad_norm": 0.0174560546875, "learning_rate": 3.720588235294118e-05, "loss": 0.4141225814819336, "step": 175 }, { "epoch": 2.6536675951717736, "grad_norm": 0.0189208984375, "learning_rate": 3.6838235294117654e-05, "loss": 0.44509315490722656, "step": 180 }, { "epoch": 2.7279480037140207, "grad_norm": 0.018310546875, "learning_rate": 3.6470588235294114e-05, "loss": 0.4210689067840576, "step": 185 }, { "epoch": 2.8022284122562673, "grad_norm": 0.0272216796875, "learning_rate": 3.610294117647059e-05, "loss": 0.4066977024078369, "step": 190 }, { "epoch": 2.8765088207985143, "grad_norm": 0.0128173828125, "learning_rate": 3.573529411764706e-05, "loss": 0.4026792049407959, "step": 195 }, { "epoch": 2.9507892293407614, "grad_norm": 0.0174560546875, "learning_rate": 3.5367647058823536e-05, "loss": 0.44449539184570314, "step": 200 } ], "logging_steps": 5, "max_steps": 680, "num_input_tokens_seen": 0, "num_train_epochs": 10, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 5.2473975207572275e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }