fasthypernet-x5 / checkpoint-200 /trainer_state.json
aixk's picture
Upload folder using huggingface_hub
4e8a296 verified
Raw
History Blame Contribute Delete
4.18 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.0027590630911508225,
"eval_steps": 500,
"global_step": 200,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.00013795315455754112,
"grad_norm": 53.4421272277832,
"learning_rate": 4.1999999999999995e-07,
"loss": 8.6569,
"step": 10
},
{
"epoch": 0.00027590630911508224,
"grad_norm": 57.65703582763672,
"learning_rate": 1.0199999999999998e-06,
"loss": 8.6391,
"step": 20
},
{
"epoch": 0.0004138594636726234,
"grad_norm": 93.79612731933594,
"learning_rate": 1.62e-06,
"loss": 8.6097,
"step": 30
},
{
"epoch": 0.0005518126182301645,
"grad_norm": 133.8379669189453,
"learning_rate": 2.22e-06,
"loss": 8.5995,
"step": 40
},
{
"epoch": 0.0006897657727877056,
"grad_norm": 99.24571228027344,
"learning_rate": 2.8199999999999997e-06,
"loss": 8.5699,
"step": 50
},
{
"epoch": 0.0008277189273452468,
"grad_norm": 155.49118041992188,
"learning_rate": 3.42e-06,
"loss": 8.5516,
"step": 60
},
{
"epoch": 0.0009656720819027878,
"grad_norm": 118.79692840576172,
"learning_rate": 4.02e-06,
"loss": 8.5331,
"step": 70
},
{
"epoch": 0.001103625236460329,
"grad_norm": 88.51729583740234,
"learning_rate": 4.62e-06,
"loss": 8.5212,
"step": 80
},
{
"epoch": 0.0012415783910178701,
"grad_norm": 117.01810455322266,
"learning_rate": 5.219999999999999e-06,
"loss": 8.5007,
"step": 90
},
{
"epoch": 0.0013795315455754113,
"grad_norm": 104.06542205810547,
"learning_rate": 5.819999999999999e-06,
"loss": 8.4948,
"step": 100
},
{
"epoch": 0.0015174847001329524,
"grad_norm": 103.02165985107422,
"learning_rate": 6.4199999999999995e-06,
"loss": 8.4871,
"step": 110
},
{
"epoch": 0.0016554378546904936,
"grad_norm": 119.96959686279297,
"learning_rate": 7.02e-06,
"loss": 8.4803,
"step": 120
},
{
"epoch": 0.0017933910092480345,
"grad_norm": 83.34615325927734,
"learning_rate": 7.619999999999999e-06,
"loss": 8.471,
"step": 130
},
{
"epoch": 0.0019313441638055756,
"grad_norm": 76.37083435058594,
"learning_rate": 8.22e-06,
"loss": 8.4655,
"step": 140
},
{
"epoch": 0.002069297318363117,
"grad_norm": 82.49830627441406,
"learning_rate": 8.819999999999999e-06,
"loss": 8.4708,
"step": 150
},
{
"epoch": 0.002207250472920658,
"grad_norm": 69.70299530029297,
"learning_rate": 9.419999999999998e-06,
"loss": 8.4659,
"step": 160
},
{
"epoch": 0.0023452036274781993,
"grad_norm": 103.55642700195312,
"learning_rate": 1.0019999999999999e-05,
"loss": 8.4555,
"step": 170
},
{
"epoch": 0.0024831567820357402,
"grad_norm": 112.4985122680664,
"learning_rate": 1.062e-05,
"loss": 8.4629,
"step": 180
},
{
"epoch": 0.002621109936593281,
"grad_norm": 78.48297882080078,
"learning_rate": 1.122e-05,
"loss": 8.4562,
"step": 190
},
{
"epoch": 0.0027590630911508225,
"grad_norm": 87.78710174560547,
"learning_rate": 1.1819999999999999e-05,
"loss": 8.451,
"step": 200
}
],
"logging_steps": 10,
"max_steps": 10000000,
"num_input_tokens_seen": 0,
"num_train_epochs": 138,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 3053619122651136.0,
"train_batch_size": 8,
"trial_name": null,
"trial_params": null
}