tourn-test-6d8b2af6 / trainer_state.json
bimabk's picture
Upload folder using huggingface_hub
9a39075 verified
Raw
History Blame Contribute Delete
5.61 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.9927536231884058,
"eval_steps": 20,
"global_step": 137,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.036231884057971016,
"grad_norm": 1.5703125,
"learning_rate": 3.139372514179859e-05,
"loss": 2.648,
"step": 5
},
{
"epoch": 0.07246376811594203,
"grad_norm": 1.6015625,
"learning_rate": 7.063588156904684e-05,
"loss": 2.5698,
"step": 10
},
{
"epoch": 0.10869565217391304,
"grad_norm": 1.328125,
"learning_rate": 9.417686158405852e-05,
"loss": 2.535,
"step": 15
},
{
"epoch": 0.14492753623188406,
"grad_norm": 1.109375,
"learning_rate": 9.412834297056845e-05,
"loss": 2.5239,
"step": 20
},
{
"epoch": 0.18115942028985507,
"grad_norm": 1.125,
"learning_rate": 9.402598775863905e-05,
"loss": 2.5369,
"step": 25
},
{
"epoch": 0.21739130434782608,
"grad_norm": 1.125,
"learning_rate": 9.386995220630276e-05,
"loss": 2.5074,
"step": 30
},
{
"epoch": 0.2536231884057971,
"grad_norm": 1.1015625,
"learning_rate": 9.366047452134543e-05,
"loss": 2.4874,
"step": 35
},
{
"epoch": 0.2898550724637681,
"grad_norm": 1.1328125,
"learning_rate": 9.339787449765238e-05,
"loss": 2.5084,
"step": 40
},
{
"epoch": 0.32608695652173914,
"grad_norm": 1.0234375,
"learning_rate": 9.308255302700287e-05,
"loss": 2.4878,
"step": 45
},
{
"epoch": 0.36231884057971014,
"grad_norm": 1.046875,
"learning_rate": 9.271499148705883e-05,
"loss": 2.4854,
"step": 50
},
{
"epoch": 0.39855072463768115,
"grad_norm": 1.0703125,
"learning_rate": 9.229575100648155e-05,
"loss": 2.501,
"step": 55
},
{
"epoch": 0.43478260869565216,
"grad_norm": 1.03125,
"learning_rate": 9.182547160829867e-05,
"loss": 2.461,
"step": 60
},
{
"epoch": 0.47101449275362317,
"grad_norm": 1.0,
"learning_rate": 9.130487123282887e-05,
"loss": 2.4792,
"step": 65
},
{
"epoch": 0.5072463768115942,
"grad_norm": 0.9921875,
"learning_rate": 9.073474464165638e-05,
"loss": 2.4696,
"step": 70
},
{
"epoch": 0.5434782608695652,
"grad_norm": 0.98046875,
"learning_rate": 9.011596220432776e-05,
"loss": 2.4758,
"step": 75
},
{
"epoch": 0.5797101449275363,
"grad_norm": 0.953125,
"learning_rate": 8.944946856962415e-05,
"loss": 2.4634,
"step": 80
},
{
"epoch": 0.6159420289855072,
"grad_norm": 0.875,
"learning_rate": 8.873628122343667e-05,
"loss": 2.4623,
"step": 85
},
{
"epoch": 0.6521739130434783,
"grad_norm": 0.94140625,
"learning_rate": 8.797748893544693e-05,
"loss": 2.4326,
"step": 90
},
{
"epoch": 0.6884057971014492,
"grad_norm": 0.93359375,
"learning_rate": 8.717425009698402e-05,
"loss": 2.4757,
"step": 95
},
{
"epoch": 0.7246376811594203,
"grad_norm": 0.953125,
"learning_rate": 8.632779095259514e-05,
"loss": 2.4531,
"step": 100
},
{
"epoch": 0.7608695652173914,
"grad_norm": 0.90625,
"learning_rate": 8.54394037280301e-05,
"loss": 2.4681,
"step": 105
},
{
"epoch": 0.7971014492753623,
"grad_norm": 0.921875,
"learning_rate": 8.45104446574969e-05,
"loss": 2.4615,
"step": 110
},
{
"epoch": 0.8333333333333334,
"grad_norm": 0.93359375,
"learning_rate": 8.35423319132008e-05,
"loss": 2.4599,
"step": 115
},
{
"epoch": 0.8695652173913043,
"grad_norm": 1.0390625,
"learning_rate": 8.253654344032692e-05,
"loss": 2.4509,
"step": 120
},
{
"epoch": 0.8695652173913043,
"eval_loss": 2.4220075607299805,
"eval_runtime": 13.5232,
"eval_samples_per_second": 14.715,
"eval_steps_per_second": 14.715,
"step": 120
},
{
"epoch": 0.9057971014492754,
"grad_norm": 1.015625,
"learning_rate": 8.149461470077207e-05,
"loss": 2.449,
"step": 125
},
{
"epoch": 0.9420289855072463,
"grad_norm": 0.84375,
"learning_rate": 8.041813632907031e-05,
"loss": 2.4368,
"step": 130
},
{
"epoch": 0.9782608695652174,
"grad_norm": 0.91015625,
"learning_rate": 7.930875170409012e-05,
"loss": 2.4394,
"step": 135
},
{
"epoch": 0.9927536231884058,
"eval_loss": 2.4152843952178955,
"eval_runtime": 12.1031,
"eval_samples_per_second": 16.442,
"eval_steps_per_second": 16.442,
"step": 137
}
],
"logging_steps": 5,
"max_steps": 414,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 20,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 1.792294366347264e+17,
"train_batch_size": 100,
"trial_name": null,
"trial_params": null
}