Jeesup's picture
muse hpsearch: muse_Llama-2-7b-hf_Books_GradDiff_lr5e-5_alpha5
3617df0 verified
Raw
History Blame Contribute Delete
6.48 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 10.0,
"eval_steps": 500,
"global_step": 180,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.28776978417266186,
"grad_norm": 117.0,
"learning_rate": 5e-05,
"loss": -4.34638671875,
"step": 5
},
{
"epoch": 0.5755395683453237,
"grad_norm": 150.0,
"learning_rate": 5e-05,
"loss": -42.646612548828124,
"step": 10
},
{
"epoch": 0.8633093525179856,
"grad_norm": 106.5,
"learning_rate": 5e-05,
"loss": -65.41669921875,
"step": 15
},
{
"epoch": 1.1151079136690647,
"grad_norm": 74.0,
"learning_rate": 5e-05,
"loss": -73.5941162109375,
"step": 20
},
{
"epoch": 1.4028776978417266,
"grad_norm": 75.0,
"learning_rate": 5e-05,
"loss": -75.83816528320312,
"step": 25
},
{
"epoch": 1.6906474820143886,
"grad_norm": 73.0,
"learning_rate": 5e-05,
"loss": -77.32380981445313,
"step": 30
},
{
"epoch": 1.9784172661870505,
"grad_norm": 73.0,
"learning_rate": 5e-05,
"loss": -78.5558349609375,
"step": 35
},
{
"epoch": 2.2302158273381294,
"grad_norm": 73.5,
"learning_rate": 5e-05,
"loss": -79.79879150390624,
"step": 40
},
{
"epoch": 2.5179856115107913,
"grad_norm": 72.5,
"learning_rate": 5e-05,
"loss": -81.05091552734375,
"step": 45
},
{
"epoch": 2.805755395683453,
"grad_norm": 75.0,
"learning_rate": 5e-05,
"loss": -82.11956176757812,
"step": 50
},
{
"epoch": 3.0575539568345325,
"grad_norm": 71.5,
"learning_rate": 5e-05,
"loss": -83.24627075195312,
"step": 55
},
{
"epoch": 3.3453237410071943,
"grad_norm": 89.5,
"learning_rate": 5e-05,
"loss": -84.156298828125,
"step": 60
},
{
"epoch": 3.633093525179856,
"grad_norm": 73.0,
"learning_rate": 5e-05,
"loss": -84.94906005859374,
"step": 65
},
{
"epoch": 3.920863309352518,
"grad_norm": 72.0,
"learning_rate": 5e-05,
"loss": -86.28233032226562,
"step": 70
},
{
"epoch": 4.172661870503597,
"grad_norm": 71.5,
"learning_rate": 5e-05,
"loss": -87.53975830078124,
"step": 75
},
{
"epoch": 4.460431654676259,
"grad_norm": 71.5,
"learning_rate": 5e-05,
"loss": -88.67244873046874,
"step": 80
},
{
"epoch": 4.748201438848921,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -89.72144165039063,
"step": 85
},
{
"epoch": 5.0,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -91.145458984375,
"step": 90
},
{
"epoch": 5.287769784172662,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -92.21284790039063,
"step": 95
},
{
"epoch": 5.575539568345324,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -93.45682983398437,
"step": 100
},
{
"epoch": 5.863309352517986,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -94.5513671875,
"step": 105
},
{
"epoch": 6.115107913669065,
"grad_norm": 71.0,
"learning_rate": 5e-05,
"loss": -95.5849609375,
"step": 110
},
{
"epoch": 6.402877697841727,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -97.01963500976562,
"step": 115
},
{
"epoch": 6.690647482014389,
"grad_norm": 72.0,
"learning_rate": 5e-05,
"loss": -98.24061889648438,
"step": 120
},
{
"epoch": 6.9784172661870505,
"grad_norm": 72.0,
"learning_rate": 5e-05,
"loss": -99.32130737304688,
"step": 125
},
{
"epoch": 7.23021582733813,
"grad_norm": 71.5,
"learning_rate": 5e-05,
"loss": -100.741357421875,
"step": 130
},
{
"epoch": 7.517985611510792,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -102.13358154296876,
"step": 135
},
{
"epoch": 7.805755395683454,
"grad_norm": 70.0,
"learning_rate": 5e-05,
"loss": -103.316259765625,
"step": 140
},
{
"epoch": 8.057553956834532,
"grad_norm": 71.5,
"learning_rate": 5e-05,
"loss": -104.76636962890625,
"step": 145
},
{
"epoch": 8.345323741007194,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -106.03206787109374,
"step": 150
},
{
"epoch": 8.633093525179856,
"grad_norm": 70.5,
"learning_rate": 5e-05,
"loss": -107.353759765625,
"step": 155
},
{
"epoch": 8.920863309352518,
"grad_norm": 69.5,
"learning_rate": 5e-05,
"loss": -108.8445556640625,
"step": 160
},
{
"epoch": 9.172661870503598,
"grad_norm": 69.5,
"learning_rate": 5e-05,
"loss": -110.38016357421876,
"step": 165
},
{
"epoch": 9.46043165467626,
"grad_norm": 69.5,
"learning_rate": 5e-05,
"loss": -111.77725830078126,
"step": 170
},
{
"epoch": 9.748201438848922,
"grad_norm": 69.5,
"learning_rate": 5e-05,
"loss": -113.11326904296875,
"step": 175
},
{
"epoch": 10.0,
"grad_norm": 70.0,
"learning_rate": 5e-05,
"loss": -114.634375,
"step": 180
},
{
"epoch": 10.0,
"step": 180,
"total_flos": 0.0,
"train_loss": -89.16345960828993,
"train_runtime": 1582.0495,
"train_samples_per_second": 3.495,
"train_steps_per_second": 0.114
}
],
"logging_steps": 5,
"max_steps": 180,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}