fasthypernet-x5 / checkpoint-400 /trainer_state.json
aixk's picture
Upload folder using huggingface_hub
7432814 verified
Raw
History Blame Contribute Delete
7.59 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.0006897657727877056,
"eval_steps": 500,
"global_step": 400,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.00013795315455754112,
"grad_norm": 53.4421272277832,
"learning_rate": 4.1999999999999995e-07,
"loss": 8.6569,
"step": 10
},
{
"epoch": 0.00027590630911508224,
"grad_norm": 57.65703582763672,
"learning_rate": 1.0199999999999998e-06,
"loss": 8.6391,
"step": 20
},
{
"epoch": 0.0004138594636726234,
"grad_norm": 93.79612731933594,
"learning_rate": 1.62e-06,
"loss": 8.6097,
"step": 30
},
{
"epoch": 0.0005518126182301645,
"grad_norm": 133.8379669189453,
"learning_rate": 2.22e-06,
"loss": 8.5995,
"step": 40
},
{
"epoch": 0.0006897657727877056,
"grad_norm": 99.24571228027344,
"learning_rate": 2.8199999999999997e-06,
"loss": 8.5699,
"step": 50
},
{
"epoch": 0.0008277189273452468,
"grad_norm": 155.49118041992188,
"learning_rate": 3.42e-06,
"loss": 8.5516,
"step": 60
},
{
"epoch": 0.0009656720819027878,
"grad_norm": 118.79692840576172,
"learning_rate": 4.02e-06,
"loss": 8.5331,
"step": 70
},
{
"epoch": 0.001103625236460329,
"grad_norm": 88.51729583740234,
"learning_rate": 4.62e-06,
"loss": 8.5212,
"step": 80
},
{
"epoch": 0.0012415783910178701,
"grad_norm": 117.01810455322266,
"learning_rate": 5.219999999999999e-06,
"loss": 8.5007,
"step": 90
},
{
"epoch": 0.0013795315455754113,
"grad_norm": 104.06542205810547,
"learning_rate": 5.819999999999999e-06,
"loss": 8.4948,
"step": 100
},
{
"epoch": 0.0015174847001329524,
"grad_norm": 103.02165985107422,
"learning_rate": 6.4199999999999995e-06,
"loss": 8.4871,
"step": 110
},
{
"epoch": 0.0016554378546904936,
"grad_norm": 119.96959686279297,
"learning_rate": 7.02e-06,
"loss": 8.4803,
"step": 120
},
{
"epoch": 0.0017933910092480345,
"grad_norm": 83.34615325927734,
"learning_rate": 7.619999999999999e-06,
"loss": 8.471,
"step": 130
},
{
"epoch": 0.0019313441638055756,
"grad_norm": 76.37083435058594,
"learning_rate": 8.22e-06,
"loss": 8.4655,
"step": 140
},
{
"epoch": 0.002069297318363117,
"grad_norm": 82.49830627441406,
"learning_rate": 8.819999999999999e-06,
"loss": 8.4708,
"step": 150
},
{
"epoch": 0.002207250472920658,
"grad_norm": 69.70299530029297,
"learning_rate": 9.419999999999998e-06,
"loss": 8.4659,
"step": 160
},
{
"epoch": 0.0023452036274781993,
"grad_norm": 103.55642700195312,
"learning_rate": 1.0019999999999999e-05,
"loss": 8.4555,
"step": 170
},
{
"epoch": 0.0024831567820357402,
"grad_norm": 112.4985122680664,
"learning_rate": 1.062e-05,
"loss": 8.4629,
"step": 180
},
{
"epoch": 0.002621109936593281,
"grad_norm": 78.48297882080078,
"learning_rate": 1.122e-05,
"loss": 8.4562,
"step": 190
},
{
"epoch": 0.0027590630911508225,
"grad_norm": 87.78710174560547,
"learning_rate": 1.1819999999999999e-05,
"loss": 8.451,
"step": 200
},
{
"epoch": 3.448828863938528e-05,
"grad_norm": 34.83774948120117,
"learning_rate": 1.2419999999999998e-05,
"loss": 8.4432,
"step": 210
},
{
"epoch": 6.897657727877056e-05,
"grad_norm": 68.05087280273438,
"learning_rate": 1.3019999999999999e-05,
"loss": 8.4542,
"step": 220
},
{
"epoch": 0.00010346486591815585,
"grad_norm": 46.86287307739258,
"learning_rate": 1.362e-05,
"loss": 8.4622,
"step": 230
},
{
"epoch": 0.00013795315455754112,
"grad_norm": 38.558746337890625,
"learning_rate": 1.4219999999999998e-05,
"loss": 8.4458,
"step": 240
},
{
"epoch": 0.0001724414431969264,
"grad_norm": 38.49115753173828,
"learning_rate": 1.4819999999999999e-05,
"loss": 8.4495,
"step": 250
},
{
"epoch": 0.0002069297318363117,
"grad_norm": 29.765907287597656,
"learning_rate": 1.5419999999999998e-05,
"loss": 8.464,
"step": 260
},
{
"epoch": 0.00024141802047569696,
"grad_norm": 48.3050537109375,
"learning_rate": 1.602e-05,
"loss": 8.4601,
"step": 270
},
{
"epoch": 0.00027590630911508224,
"grad_norm": 27.089998245239258,
"learning_rate": 1.6619999999999997e-05,
"loss": 8.4628,
"step": 280
},
{
"epoch": 0.00031039459775446753,
"grad_norm": 36.98786163330078,
"learning_rate": 1.7219999999999998e-05,
"loss": 8.4364,
"step": 290
},
{
"epoch": 0.0003448828863938528,
"grad_norm": 39.274784088134766,
"learning_rate": 1.782e-05,
"loss": 8.4314,
"step": 300
},
{
"epoch": 0.0003793711750332381,
"grad_norm": 29.303476333618164,
"learning_rate": 1.842e-05,
"loss": 8.4153,
"step": 310
},
{
"epoch": 0.0004138594636726234,
"grad_norm": 24.435274124145508,
"learning_rate": 1.9019999999999997e-05,
"loss": 8.4489,
"step": 320
},
{
"epoch": 0.0004483477523120086,
"grad_norm": 25.8426456451416,
"learning_rate": 1.962e-05,
"loss": 8.4297,
"step": 330
},
{
"epoch": 0.0004828360409513939,
"grad_norm": 24.164419174194336,
"learning_rate": 2.022e-05,
"loss": 8.4398,
"step": 340
},
{
"epoch": 0.0005173243295907793,
"grad_norm": 17.73716163635254,
"learning_rate": 2.082e-05,
"loss": 8.4333,
"step": 350
},
{
"epoch": 0.0005518126182301645,
"grad_norm": 34.02337646484375,
"learning_rate": 2.1419999999999998e-05,
"loss": 8.4488,
"step": 360
},
{
"epoch": 0.0005863009068695498,
"grad_norm": 26.535123825073242,
"learning_rate": 2.202e-05,
"loss": 8.4361,
"step": 370
},
{
"epoch": 0.0006207891955089351,
"grad_norm": 32.27824783325195,
"learning_rate": 2.2619999999999997e-05,
"loss": 8.4116,
"step": 380
},
{
"epoch": 0.0006552774841483203,
"grad_norm": 20.55720329284668,
"learning_rate": 2.3219999999999998e-05,
"loss": 8.4151,
"step": 390
},
{
"epoch": 0.0006897657727877056,
"grad_norm": 30.87930679321289,
"learning_rate": 2.382e-05,
"loss": 8.43,
"step": 400
}
],
"logging_steps": 10,
"max_steps": 10000000,
"num_input_tokens_seen": 0,
"num_train_epochs": 35,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 3845245063495680.0,
"train_batch_size": 8,
"trial_name": null,
"trial_params": null
}