Jeesup's picture
muse exact: muse_Llama-2-7b-hf_Books_GradDiff_lr1e-5
a7c3a3f verified
Raw
History Blame Contribute Delete
6.53 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 10.0,
"eval_steps": 500,
"global_step": 180,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.28776978417266186,
"grad_norm": 3.625,
"learning_rate": 1e-05,
"loss": -0.8489187240600586,
"step": 5
},
{
"epoch": 0.5755395683453237,
"grad_norm": 6.625,
"learning_rate": 1e-05,
"loss": -1.094129753112793,
"step": 10
},
{
"epoch": 0.8633093525179856,
"grad_norm": 6.4375,
"learning_rate": 1e-05,
"loss": -1.3786694526672363,
"step": 15
},
{
"epoch": 1.1151079136690647,
"grad_norm": 11.25,
"learning_rate": 1e-05,
"loss": -2.602250099182129,
"step": 20
},
{
"epoch": 1.4028776978417266,
"grad_norm": 21.125,
"learning_rate": 1e-05,
"loss": -2.348813056945801,
"step": 25
},
{
"epoch": 1.6906474820143886,
"grad_norm": 98.5,
"learning_rate": 1e-05,
"loss": -3.6698333740234377,
"step": 30
},
{
"epoch": 1.9784172661870505,
"grad_norm": 247.0,
"learning_rate": 1e-05,
"loss": -11.704106903076172,
"step": 35
},
{
"epoch": 2.2302158273381294,
"grad_norm": 908.0,
"learning_rate": 1e-05,
"loss": -24.895542907714844,
"step": 40
},
{
"epoch": 2.5179856115107913,
"grad_norm": 170.0,
"learning_rate": 1e-05,
"loss": -38.253057861328124,
"step": 45
},
{
"epoch": 2.805755395683453,
"grad_norm": 128.0,
"learning_rate": 1e-05,
"loss": -49.457061767578125,
"step": 50
},
{
"epoch": 3.0575539568345325,
"grad_norm": 163.0,
"learning_rate": 1e-05,
"loss": -60.77360229492187,
"step": 55
},
{
"epoch": 3.3453237410071943,
"grad_norm": 180.0,
"learning_rate": 1e-05,
"loss": -77.50135498046875,
"step": 60
},
{
"epoch": 3.633093525179856,
"grad_norm": 153.0,
"learning_rate": 1e-05,
"loss": -96.42343139648438,
"step": 65
},
{
"epoch": 3.920863309352518,
"grad_norm": 119.0,
"learning_rate": 1e-05,
"loss": -112.23983154296874,
"step": 70
},
{
"epoch": 4.172661870503597,
"grad_norm": 123.0,
"learning_rate": 1e-05,
"loss": -123.47835693359374,
"step": 75
},
{
"epoch": 4.460431654676259,
"grad_norm": 138.0,
"learning_rate": 1e-05,
"loss": -130.45103759765624,
"step": 80
},
{
"epoch": 4.748201438848921,
"grad_norm": 106.5,
"learning_rate": 1e-05,
"loss": -134.79713134765626,
"step": 85
},
{
"epoch": 5.0,
"grad_norm": 117.0,
"learning_rate": 1e-05,
"loss": -136.9400634765625,
"step": 90
},
{
"epoch": 5.287769784172662,
"grad_norm": 90.0,
"learning_rate": 1e-05,
"loss": -138.45052490234374,
"step": 95
},
{
"epoch": 5.575539568345324,
"grad_norm": 90.5,
"learning_rate": 1e-05,
"loss": -139.13138427734376,
"step": 100
},
{
"epoch": 5.863309352517986,
"grad_norm": 85.5,
"learning_rate": 1e-05,
"loss": -139.83267822265626,
"step": 105
},
{
"epoch": 6.115107913669065,
"grad_norm": 81.5,
"learning_rate": 1e-05,
"loss": -140.0307373046875,
"step": 110
},
{
"epoch": 6.402877697841727,
"grad_norm": 80.5,
"learning_rate": 1e-05,
"loss": -140.39393310546876,
"step": 115
},
{
"epoch": 6.690647482014389,
"grad_norm": 81.0,
"learning_rate": 1e-05,
"loss": -140.50172119140626,
"step": 120
},
{
"epoch": 6.9784172661870505,
"grad_norm": 80.5,
"learning_rate": 1e-05,
"loss": -140.62662353515626,
"step": 125
},
{
"epoch": 7.23021582733813,
"grad_norm": 83.0,
"learning_rate": 1e-05,
"loss": -140.66796875,
"step": 130
},
{
"epoch": 7.517985611510792,
"grad_norm": 81.5,
"learning_rate": 1e-05,
"loss": -140.8075439453125,
"step": 135
},
{
"epoch": 7.805755395683454,
"grad_norm": 78.5,
"learning_rate": 1e-05,
"loss": -140.95115966796874,
"step": 140
},
{
"epoch": 8.057553956834532,
"grad_norm": 78.5,
"learning_rate": 1e-05,
"loss": -140.95902099609376,
"step": 145
},
{
"epoch": 8.345323741007194,
"grad_norm": 77.5,
"learning_rate": 1e-05,
"loss": -141.0729736328125,
"step": 150
},
{
"epoch": 8.633093525179856,
"grad_norm": 79.0,
"learning_rate": 1e-05,
"loss": -141.2009765625,
"step": 155
},
{
"epoch": 8.920863309352518,
"grad_norm": 79.0,
"learning_rate": 1e-05,
"loss": -141.22720947265626,
"step": 160
},
{
"epoch": 9.172661870503598,
"grad_norm": 77.5,
"learning_rate": 1e-05,
"loss": -141.30667724609376,
"step": 165
},
{
"epoch": 9.46043165467626,
"grad_norm": 77.5,
"learning_rate": 1e-05,
"loss": -141.33355712890625,
"step": 170
},
{
"epoch": 9.748201438848922,
"grad_norm": 77.0,
"learning_rate": 1e-05,
"loss": -141.43316650390625,
"step": 175
},
{
"epoch": 10.0,
"grad_norm": 80.5,
"learning_rate": 1e-05,
"loss": -141.4542236328125,
"step": 180
},
{
"epoch": 10.0,
"step": 180,
"total_flos": 0.0,
"train_loss": -98.33997982078128,
"train_runtime": 1576.0815,
"train_samples_per_second": 3.509,
"train_steps_per_second": 0.114
}
],
"logging_steps": 5,
"max_steps": 180,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}