fasthypernet-12 / checkpoint-400 /trainer_state.json
aixk's picture
Upload folder using huggingface_hub
f28f969 verified
Raw
History Blame Contribute Delete
7.51 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.08399391044149299,
"eval_steps": 500,
"global_step": 400,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.002099847761037325,
"grad_norm": 19872.2890625,
"learning_rate": 3e-07,
"loss": 8.7939,
"step": 10
},
{
"epoch": 0.00419969552207465,
"grad_norm": 1458.198974609375,
"learning_rate": 1.2999999999999998e-06,
"loss": 8.1117,
"step": 20
},
{
"epoch": 0.006299543283111974,
"grad_norm": 297.6234130859375,
"learning_rate": 2.2999999999999996e-06,
"loss": 7.3977,
"step": 30
},
{
"epoch": 0.0083993910441493,
"grad_norm": 426.92987060546875,
"learning_rate": 3.2999999999999993e-06,
"loss": 7.2069,
"step": 40
},
{
"epoch": 0.010499238805186623,
"grad_norm": 474.7532653808594,
"learning_rate": 4.2999999999999995e-06,
"loss": 7.08,
"step": 50
},
{
"epoch": 0.012599086566223949,
"grad_norm": 1569.4381103515625,
"learning_rate": 5.3e-06,
"loss": 7.0092,
"step": 60
},
{
"epoch": 0.014698934327261274,
"grad_norm": 1407.21142578125,
"learning_rate": 6.3e-06,
"loss": 6.9929,
"step": 70
},
{
"epoch": 0.0167987820882986,
"grad_norm": 500.49493408203125,
"learning_rate": 7.299999999999999e-06,
"loss": 6.9501,
"step": 80
},
{
"epoch": 0.018898629849335925,
"grad_norm": 442.69537353515625,
"learning_rate": 8.299999999999998e-06,
"loss": 6.9001,
"step": 90
},
{
"epoch": 0.020998477610373247,
"grad_norm": 1423.2950439453125,
"learning_rate": 9.299999999999999e-06,
"loss": 6.9037,
"step": 100
},
{
"epoch": 0.023098325371410572,
"grad_norm": 448.7774963378906,
"learning_rate": 1.03e-05,
"loss": 6.8732,
"step": 110
},
{
"epoch": 0.025198173132447897,
"grad_norm": 711.8992919921875,
"learning_rate": 1.1299999999999999e-05,
"loss": 6.8381,
"step": 120
},
{
"epoch": 0.027298020893485223,
"grad_norm": 884.006591796875,
"learning_rate": 1.2299999999999999e-05,
"loss": 6.8462,
"step": 130
},
{
"epoch": 0.029397868654522548,
"grad_norm": 96.5859375,
"learning_rate": 1.33e-05,
"loss": 6.8433,
"step": 140
},
{
"epoch": 0.03149771641555987,
"grad_norm": 91.52490997314453,
"learning_rate": 1.43e-05,
"loss": 6.8171,
"step": 150
},
{
"epoch": 0.0335975641765972,
"grad_norm": 354.9374084472656,
"learning_rate": 1.53e-05,
"loss": 6.7899,
"step": 160
},
{
"epoch": 0.035697411937634524,
"grad_norm": 189.09506225585938,
"learning_rate": 1.6299999999999996e-05,
"loss": 6.7456,
"step": 170
},
{
"epoch": 0.03779725969867185,
"grad_norm": 170.39376831054688,
"learning_rate": 1.7299999999999997e-05,
"loss": 6.7505,
"step": 180
},
{
"epoch": 0.03989710745970917,
"grad_norm": 193.9166259765625,
"learning_rate": 1.8299999999999998e-05,
"loss": 6.752,
"step": 190
},
{
"epoch": 0.04199695522074649,
"grad_norm": 336.08624267578125,
"learning_rate": 1.93e-05,
"loss": 6.7015,
"step": 200
},
{
"epoch": 0.04409680298178382,
"grad_norm": 307.40557861328125,
"learning_rate": 2.03e-05,
"loss": 6.7024,
"step": 210
},
{
"epoch": 0.046196650742821144,
"grad_norm": 175.94061279296875,
"learning_rate": 2.1299999999999996e-05,
"loss": 6.6639,
"step": 220
},
{
"epoch": 0.04829649850385847,
"grad_norm": 496.3529052734375,
"learning_rate": 2.23e-05,
"loss": 6.6309,
"step": 230
},
{
"epoch": 0.050396346264895794,
"grad_norm": 112.63114929199219,
"learning_rate": 2.3299999999999997e-05,
"loss": 6.6146,
"step": 240
},
{
"epoch": 0.05249619402593312,
"grad_norm": 141.9170379638672,
"learning_rate": 2.4299999999999998e-05,
"loss": 6.6386,
"step": 250
},
{
"epoch": 0.054596041786970445,
"grad_norm": 120.92185974121094,
"learning_rate": 2.53e-05,
"loss": 6.6323,
"step": 260
},
{
"epoch": 0.05669588954800777,
"grad_norm": 241.84129333496094,
"learning_rate": 2.63e-05,
"loss": 6.598,
"step": 270
},
{
"epoch": 0.058795737309045096,
"grad_norm": 115.57807159423828,
"learning_rate": 2.7299999999999996e-05,
"loss": 6.5549,
"step": 280
},
{
"epoch": 0.06089558507008242,
"grad_norm": 99.67350769042969,
"learning_rate": 2.83e-05,
"loss": 6.5484,
"step": 290
},
{
"epoch": 0.06299543283111975,
"grad_norm": 105.20110321044922,
"learning_rate": 2.9299999999999997e-05,
"loss": 6.5416,
"step": 300
},
{
"epoch": 0.06509528059215706,
"grad_norm": 207.86203002929688,
"learning_rate": 3.0299999999999998e-05,
"loss": 6.5463,
"step": 310
},
{
"epoch": 0.0671951283531944,
"grad_norm": 125.6773910522461,
"learning_rate": 3.1299999999999995e-05,
"loss": 6.4848,
"step": 320
},
{
"epoch": 0.06929497611423172,
"grad_norm": 121.31608581542969,
"learning_rate": 3.229999999999999e-05,
"loss": 6.4763,
"step": 330
},
{
"epoch": 0.07139482387526905,
"grad_norm": 86.57124328613281,
"learning_rate": 3.3299999999999996e-05,
"loss": 6.4579,
"step": 340
},
{
"epoch": 0.07349467163630637,
"grad_norm": 105.74996948242188,
"learning_rate": 3.4299999999999993e-05,
"loss": 6.408,
"step": 350
},
{
"epoch": 0.0755945193973437,
"grad_norm": 68.95454406738281,
"learning_rate": 3.53e-05,
"loss": 6.3928,
"step": 360
},
{
"epoch": 0.07769436715838102,
"grad_norm": 107.94305419921875,
"learning_rate": 3.6299999999999995e-05,
"loss": 6.3853,
"step": 370
},
{
"epoch": 0.07979421491941834,
"grad_norm": 136.39556884765625,
"learning_rate": 3.73e-05,
"loss": 6.3639,
"step": 380
},
{
"epoch": 0.08189406268045567,
"grad_norm": 152.05712890625,
"learning_rate": 3.83e-05,
"loss": 6.348,
"step": 390
},
{
"epoch": 0.08399391044149299,
"grad_norm": 139.25466918945312,
"learning_rate": 3.93e-05,
"loss": 6.3261,
"step": 400
}
],
"logging_steps": 10,
"max_steps": 100000,
"num_input_tokens_seen": 0,
"num_train_epochs": 21,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.60705846759424e+16,
"train_batch_size": 16,
"trial_name": null,
"trial_params": null
}