{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 782, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0256, "grad_norm": 0.6669269800186157, "learning_rate": 1.9e-05, "loss": 8.179115295410156, "step": 20 }, { "epoch": 0.0512, "grad_norm": 0.5379630327224731, "learning_rate": 3.9000000000000006e-05, "loss": 8.11802978515625, "step": 40 }, { "epoch": 0.0768, "grad_norm": 0.5859383940696716, "learning_rate": 5.9e-05, "loss": 8.041471862792969, "step": 60 }, { "epoch": 0.1024, "grad_norm": 0.6329635381698608, "learning_rate": 7.900000000000001e-05, "loss": 7.967082977294922, "step": 80 }, { "epoch": 0.128, "grad_norm": 0.7737689018249512, "learning_rate": 9.900000000000001e-05, "loss": 7.894486999511718, "step": 100 }, { "epoch": 0.1536, "grad_norm": 1.0059387683868408, "learning_rate": 9.721407624633431e-05, "loss": 7.814192199707032, "step": 120 }, { "epoch": 0.1792, "grad_norm": 0.804948091506958, "learning_rate": 9.428152492668623e-05, "loss": 7.729258728027344, "step": 140 }, { "epoch": 0.2048, "grad_norm": 0.9692172408103943, "learning_rate": 9.134897360703812e-05, "loss": 7.668946838378906, "step": 160 }, { "epoch": 0.2304, "grad_norm": 0.9547553062438965, "learning_rate": 8.841642228739004e-05, "loss": 7.622274780273438, "step": 180 }, { "epoch": 0.256, "grad_norm": 0.6002941727638245, "learning_rate": 8.548387096774195e-05, "loss": 7.579273986816406, "step": 200 }, { "epoch": 0.2816, "grad_norm": 0.8823673725128174, "learning_rate": 8.255131964809384e-05, "loss": 7.519114685058594, "step": 220 }, { "epoch": 0.3072, "grad_norm": 1.0416988134384155, "learning_rate": 7.961876832844574e-05, "loss": 7.484053039550782, "step": 240 }, { "epoch": 0.3328, "grad_norm": 0.735251784324646, "learning_rate": 7.668621700879765e-05, "loss": 7.443171691894531, "step": 260 }, { "epoch": 0.3584, "grad_norm": 1.0857356786727905, "learning_rate": 7.375366568914957e-05, "loss": 7.403555297851563, "step": 280 }, { "epoch": 0.384, "grad_norm": 0.9961302280426025, "learning_rate": 7.082111436950148e-05, "loss": 7.371603393554688, "step": 300 }, { "epoch": 0.4096, "grad_norm": 0.8889548182487488, "learning_rate": 6.788856304985338e-05, "loss": 7.352679443359375, "step": 320 }, { "epoch": 0.4352, "grad_norm": 1.0873897075653076, "learning_rate": 6.495601173020527e-05, "loss": 7.3021392822265625, "step": 340 }, { "epoch": 0.4608, "grad_norm": 0.7345384359359741, "learning_rate": 6.202346041055719e-05, "loss": 7.2823539733886715, "step": 360 }, { "epoch": 0.4864, "grad_norm": 0.6853398084640503, "learning_rate": 5.90909090909091e-05, "loss": 7.260960388183594, "step": 380 }, { "epoch": 0.512, "grad_norm": 0.9110084772109985, "learning_rate": 5.6158357771260995e-05, "loss": 7.228886413574219, "step": 400 }, { "epoch": 0.5376, "grad_norm": 1.127113938331604, "learning_rate": 5.32258064516129e-05, "loss": 7.202789306640625, "step": 420 }, { "epoch": 0.5632, "grad_norm": 0.7524133324623108, "learning_rate": 5.029325513196481e-05, "loss": 7.181660461425781, "step": 440 }, { "epoch": 0.5888, "grad_norm": 0.7863721251487732, "learning_rate": 4.736070381231672e-05, "loss": 7.178968048095703, "step": 460 }, { "epoch": 0.6144, "grad_norm": 0.9041445851325989, "learning_rate": 4.442815249266862e-05, "loss": 7.156215667724609, "step": 480 }, { "epoch": 0.64, "grad_norm": 0.9909660220146179, "learning_rate": 4.149560117302053e-05, "loss": 7.125262451171875, "step": 500 }, { "epoch": 0.6656, "grad_norm": 0.816728949546814, "learning_rate": 3.856304985337244e-05, "loss": 7.123587036132813, "step": 520 }, { "epoch": 0.6912, "grad_norm": 0.8992451429367065, "learning_rate": 3.563049853372434e-05, "loss": 7.094441223144531, "step": 540 }, { "epoch": 0.7168, "grad_norm": 1.083111047744751, "learning_rate": 3.269794721407625e-05, "loss": 7.08851318359375, "step": 560 }, { "epoch": 0.7424, "grad_norm": 0.7089452147483826, "learning_rate": 2.9765395894428155e-05, "loss": 7.087593841552734, "step": 580 }, { "epoch": 0.768, "grad_norm": 0.7186883687973022, "learning_rate": 2.683284457478006e-05, "loss": 7.07503662109375, "step": 600 }, { "epoch": 0.7936, "grad_norm": 0.8105409741401672, "learning_rate": 2.3900293255131968e-05, "loss": 7.063044738769531, "step": 620 }, { "epoch": 0.8192, "grad_norm": 0.6952916979789734, "learning_rate": 2.0967741935483873e-05, "loss": 7.058567047119141, "step": 640 }, { "epoch": 0.8448, "grad_norm": 0.7389153838157654, "learning_rate": 1.8035190615835778e-05, "loss": 7.051625061035156, "step": 660 }, { "epoch": 0.8704, "grad_norm": 0.7044821381568909, "learning_rate": 1.5102639296187684e-05, "loss": 7.039373779296875, "step": 680 }, { "epoch": 0.896, "grad_norm": 0.7552204728126526, "learning_rate": 1.217008797653959e-05, "loss": 7.039744567871094, "step": 700 }, { "epoch": 0.9216, "grad_norm": 0.8664493560791016, "learning_rate": 9.237536656891495e-06, "loss": 7.020960998535156, "step": 720 }, { "epoch": 0.9472, "grad_norm": 0.5861105918884277, "learning_rate": 6.304985337243402e-06, "loss": 7.012393188476563, "step": 740 }, { "epoch": 0.9728, "grad_norm": 0.7869713306427002, "learning_rate": 3.372434017595308e-06, "loss": 7.024510192871094, "step": 760 }, { "epoch": 0.9984, "grad_norm": 0.6115875840187073, "learning_rate": 4.3988269794721416e-07, "loss": 7.033552551269532, "step": 780 } ], "logging_steps": 20, "max_steps": 782, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 259881610772480.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }