fasthypernet-fs83 / checkpoint-300 /trainer_state.json
aixk's picture
Upload folder using huggingface_hub
0dc75c5 verified
Raw
History Blame Contribute Delete
5.72 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.012767994892802044,
"eval_steps": 500,
"global_step": 300,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.0004255998297600681,
"grad_norm": 5634.16845703125,
"learning_rate": 2.4e-07,
"loss": 14.1264,
"step": 10
},
{
"epoch": 0.0008511996595201362,
"grad_norm": 2312.5771484375,
"learning_rate": 1.04e-06,
"loss": 13.2905,
"step": 20
},
{
"epoch": 0.0012767994892802044,
"grad_norm": 2110.1591796875,
"learning_rate": 1.84e-06,
"loss": 12.6236,
"step": 30
},
{
"epoch": 0.0017023993190402723,
"grad_norm": 1493.8431396484375,
"learning_rate": 2.64e-06,
"loss": 12.3954,
"step": 40
},
{
"epoch": 0.0021279991488003403,
"grad_norm": 1430.7598876953125,
"learning_rate": 3.44e-06,
"loss": 12.265,
"step": 50
},
{
"epoch": 0.0025535989785604087,
"grad_norm": 2192.621826171875,
"learning_rate": 4.24e-06,
"loss": 12.018,
"step": 60
},
{
"epoch": 0.0029791988083204767,
"grad_norm": 1678.5159912109375,
"learning_rate": 5.04e-06,
"loss": 11.8932,
"step": 70
},
{
"epoch": 0.0034047986380805447,
"grad_norm": 1261.883056640625,
"learning_rate": 5.84e-06,
"loss": 11.8077,
"step": 80
},
{
"epoch": 0.0038303984678406127,
"grad_norm": 1710.3070068359375,
"learning_rate": 6.640000000000001e-06,
"loss": 11.6841,
"step": 90
},
{
"epoch": 0.004255998297600681,
"grad_norm": 2306.063232421875,
"learning_rate": 7.44e-06,
"loss": 11.5426,
"step": 100
},
{
"epoch": 0.004681598127360749,
"grad_norm": 2158.360107421875,
"learning_rate": 8.24e-06,
"loss": 11.4863,
"step": 110
},
{
"epoch": 0.0051071979571208174,
"grad_norm": 2058.938232421875,
"learning_rate": 9.04e-06,
"loss": 11.5044,
"step": 120
},
{
"epoch": 0.005532797786880885,
"grad_norm": 2320.1376953125,
"learning_rate": 9.84e-06,
"loss": 11.4696,
"step": 130
},
{
"epoch": 0.005958397616640953,
"grad_norm": 3147.455078125,
"learning_rate": 1.064e-05,
"loss": 11.2994,
"step": 140
},
{
"epoch": 0.006383997446401022,
"grad_norm": 1930.887451171875,
"learning_rate": 1.144e-05,
"loss": 11.2686,
"step": 150
},
{
"epoch": 0.006809597276161089,
"grad_norm": 3005.495361328125,
"learning_rate": 1.224e-05,
"loss": 11.2556,
"step": 160
},
{
"epoch": 0.007235197105921158,
"grad_norm": 1587.324462890625,
"learning_rate": 1.3039999999999999e-05,
"loss": 11.1593,
"step": 170
},
{
"epoch": 0.007660796935681225,
"grad_norm": 1496.0810546875,
"learning_rate": 1.384e-05,
"loss": 11.0329,
"step": 180
},
{
"epoch": 0.008086396765441295,
"grad_norm": 2397.255615234375,
"learning_rate": 1.4560000000000001e-05,
"loss": 11.0514,
"step": 190
},
{
"epoch": 0.008511996595201361,
"grad_norm": 3627.1494140625,
"learning_rate": 1.536e-05,
"loss": 10.9684,
"step": 200
},
{
"epoch": 0.00893759642496143,
"grad_norm": 1078.9390869140625,
"learning_rate": 1.616e-05,
"loss": 11.0077,
"step": 210
},
{
"epoch": 0.009363196254721498,
"grad_norm": 1400.2003173828125,
"learning_rate": 1.696e-05,
"loss": 10.7138,
"step": 220
},
{
"epoch": 0.009788796084481566,
"grad_norm": 880.6416015625,
"learning_rate": 1.7760000000000003e-05,
"loss": 10.4564,
"step": 230
},
{
"epoch": 0.010214395914241635,
"grad_norm": 888.9507446289062,
"learning_rate": 1.856e-05,
"loss": 10.4699,
"step": 240
},
{
"epoch": 0.010639995744001702,
"grad_norm": 604.7562255859375,
"learning_rate": 1.936e-05,
"loss": 10.5335,
"step": 250
},
{
"epoch": 0.01106559557376177,
"grad_norm": 885.2161865234375,
"learning_rate": 2.016e-05,
"loss": 10.3258,
"step": 260
},
{
"epoch": 0.011491195403521838,
"grad_norm": 689.9547729492188,
"learning_rate": 2.0960000000000003e-05,
"loss": 10.3034,
"step": 270
},
{
"epoch": 0.011916795233281907,
"grad_norm": 1086.3548583984375,
"learning_rate": 2.176e-05,
"loss": 10.121,
"step": 280
},
{
"epoch": 0.012342395063041975,
"grad_norm": 728.9247436523438,
"learning_rate": 2.256e-05,
"loss": 9.9983,
"step": 290
},
{
"epoch": 0.012767994892802044,
"grad_norm": 931.4705200195312,
"learning_rate": 2.336e-05,
"loss": 10.1502,
"step": 300
}
],
"logging_steps": 10,
"max_steps": 10000000,
"num_input_tokens_seen": 0,
"num_train_epochs": 426,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 4405008815308800.0,
"train_batch_size": 8,
"trial_name": null,
"trial_params": null
}