{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.8714596949891068, "eval_steps": 500, "global_step": 4000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.02178649237472767, "grad_norm": 1.472805380821228, "learning_rate": 4.945533769063181e-05, "loss": 0.6911, "step": 100 }, { "epoch": 0.04357298474945534, "grad_norm": 1.3399823904037476, "learning_rate": 4.891067538126362e-05, "loss": 0.4494, "step": 200 }, { "epoch": 0.06535947712418301, "grad_norm": 1.9283204078674316, "learning_rate": 4.8366013071895424e-05, "loss": 0.4761, "step": 300 }, { "epoch": 0.08714596949891068, "grad_norm": 3.021352529525757, "learning_rate": 4.7821350762527234e-05, "loss": 0.4803, "step": 400 }, { "epoch": 0.10893246187363835, "grad_norm": 0.3251399099826813, "learning_rate": 4.7276688453159044e-05, "loss": 0.438, "step": 500 }, { "epoch": 0.13071895424836602, "grad_norm": 1.289565920829773, "learning_rate": 4.673202614379085e-05, "loss": 0.4677, "step": 600 }, { "epoch": 0.15250544662309368, "grad_norm": 1.5132582187652588, "learning_rate": 4.6187363834422656e-05, "loss": 0.5276, "step": 700 }, { "epoch": 0.17429193899782136, "grad_norm": 0.4088970720767975, "learning_rate": 4.564270152505447e-05, "loss": 0.444, "step": 800 }, { "epoch": 0.19607843137254902, "grad_norm": 2.157912015914917, "learning_rate": 4.5098039215686275e-05, "loss": 0.4502, "step": 900 }, { "epoch": 0.2178649237472767, "grad_norm": 1.5439516305923462, "learning_rate": 4.4553376906318085e-05, "loss": 0.4248, "step": 1000 }, { "epoch": 0.23965141612200436, "grad_norm": 1.3628387451171875, "learning_rate": 4.400871459694989e-05, "loss": 0.4223, "step": 1100 }, { "epoch": 0.26143790849673204, "grad_norm": 1.5868219137191772, "learning_rate": 4.3464052287581704e-05, "loss": 0.4877, "step": 1200 }, { "epoch": 0.28322440087145967, "grad_norm": 1.0417579412460327, "learning_rate": 4.291938997821351e-05, "loss": 0.4705, "step": 1300 }, { "epoch": 0.30501089324618735, "grad_norm": 1.6881535053253174, "learning_rate": 4.2374727668845316e-05, "loss": 0.4305, "step": 1400 }, { "epoch": 0.32679738562091504, "grad_norm": 1.3759115934371948, "learning_rate": 4.1830065359477126e-05, "loss": 0.3975, "step": 1500 }, { "epoch": 0.3485838779956427, "grad_norm": 2.2703583240509033, "learning_rate": 4.1285403050108935e-05, "loss": 0.4246, "step": 1600 }, { "epoch": 0.37037037037037035, "grad_norm": 1.0483826398849487, "learning_rate": 4.074074074074074e-05, "loss": 0.4994, "step": 1700 }, { "epoch": 0.39215686274509803, "grad_norm": 1.2435953617095947, "learning_rate": 4.0196078431372555e-05, "loss": 0.4337, "step": 1800 }, { "epoch": 0.4139433551198257, "grad_norm": 1.4106743335723877, "learning_rate": 3.965141612200436e-05, "loss": 0.4338, "step": 1900 }, { "epoch": 0.4357298474945534, "grad_norm": 0.9341331720352173, "learning_rate": 3.910675381263617e-05, "loss": 0.4683, "step": 2000 }, { "epoch": 0.45751633986928103, "grad_norm": 1.9387885332107544, "learning_rate": 3.8562091503267977e-05, "loss": 0.4207, "step": 2100 }, { "epoch": 0.4793028322440087, "grad_norm": 0.12210940569639206, "learning_rate": 3.8017429193899786e-05, "loss": 0.4549, "step": 2200 }, { "epoch": 0.5010893246187363, "grad_norm": 1.2780179977416992, "learning_rate": 3.747276688453159e-05, "loss": 0.4325, "step": 2300 }, { "epoch": 0.5228758169934641, "grad_norm": 0.6933521032333374, "learning_rate": 3.6928104575163405e-05, "loss": 0.407, "step": 2400 }, { "epoch": 0.5446623093681917, "grad_norm": 1.5122122764587402, "learning_rate": 3.638344226579521e-05, "loss": 0.4629, "step": 2500 }, { "epoch": 0.5664488017429193, "grad_norm": 0.967708170413971, "learning_rate": 3.583877995642702e-05, "loss": 0.4102, "step": 2600 }, { "epoch": 0.5882352941176471, "grad_norm": 2.7059736251831055, "learning_rate": 3.529411764705883e-05, "loss": 0.4465, "step": 2700 }, { "epoch": 0.6100217864923747, "grad_norm": 1.3728137016296387, "learning_rate": 3.474945533769064e-05, "loss": 0.4225, "step": 2800 }, { "epoch": 0.6318082788671024, "grad_norm": 0.8513820767402649, "learning_rate": 3.420479302832244e-05, "loss": 0.3922, "step": 2900 }, { "epoch": 0.6535947712418301, "grad_norm": 1.2829489707946777, "learning_rate": 3.366013071895425e-05, "loss": 0.4202, "step": 3000 }, { "epoch": 0.6753812636165577, "grad_norm": 1.9668824672698975, "learning_rate": 3.311546840958606e-05, "loss": 0.4425, "step": 3100 }, { "epoch": 0.6971677559912854, "grad_norm": 0.9782924056053162, "learning_rate": 3.257080610021787e-05, "loss": 0.4407, "step": 3200 }, { "epoch": 0.7189542483660131, "grad_norm": 1.2658343315124512, "learning_rate": 3.202614379084967e-05, "loss": 0.4256, "step": 3300 }, { "epoch": 0.7407407407407407, "grad_norm": 1.1413471698760986, "learning_rate": 3.148148148148148e-05, "loss": 0.435, "step": 3400 }, { "epoch": 0.7625272331154684, "grad_norm": 0.6810837984085083, "learning_rate": 3.093681917211329e-05, "loss": 0.4662, "step": 3500 }, { "epoch": 0.7843137254901961, "grad_norm": 1.4281896352767944, "learning_rate": 3.0392156862745097e-05, "loss": 0.4148, "step": 3600 }, { "epoch": 0.8061002178649237, "grad_norm": 0.907742977142334, "learning_rate": 2.984749455337691e-05, "loss": 0.417, "step": 3700 }, { "epoch": 0.8278867102396514, "grad_norm": 1.6939854621887207, "learning_rate": 2.9302832244008716e-05, "loss": 0.3971, "step": 3800 }, { "epoch": 0.8496732026143791, "grad_norm": 2.122053384780884, "learning_rate": 2.8758169934640522e-05, "loss": 0.416, "step": 3900 }, { "epoch": 0.8714596949891068, "grad_norm": 2.520594596862793, "learning_rate": 2.8213507625272335e-05, "loss": 0.4205, "step": 4000 } ], "logging_steps": 100, "max_steps": 9180, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 1000, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 3.478218866688e+16, "train_batch_size": 8, "trial_name": null, "trial_params": null }