code-llama-sft / checkpoint-500 /trainer_state.json
aq1048576's picture
Upload code SFT checkpoint
99a0cea verified
Raw
History Blame Contribute Delete
5.1 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.9086778736937755,
"eval_steps": 500,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.03634711494775102,
"grad_norm": 1.1015625,
"learning_rate": 4.997919498361457e-05,
"loss": 1.0851,
"step": 20
},
{
"epoch": 0.07269422989550205,
"grad_norm": 0.90625,
"learning_rate": 4.969106773218577e-05,
"loss": 1.017,
"step": 40
},
{
"epoch": 0.10904134484325306,
"grad_norm": 0.9375,
"learning_rate": 4.906779742499551e-05,
"loss": 0.9936,
"step": 60
},
{
"epoch": 0.1453884597910041,
"grad_norm": 0.80078125,
"learning_rate": 4.811784399247625e-05,
"loss": 0.9768,
"step": 80
},
{
"epoch": 0.18173557473875512,
"grad_norm": 0.75,
"learning_rate": 4.685410158321884e-05,
"loss": 0.9572,
"step": 100
},
{
"epoch": 0.21808268968650613,
"grad_norm": 0.83984375,
"learning_rate": 4.5293723545848285e-05,
"loss": 0.9608,
"step": 120
},
{
"epoch": 0.25442980463425713,
"grad_norm": 0.71875,
"learning_rate": 4.345788959884658e-05,
"loss": 0.9339,
"step": 140
},
{
"epoch": 0.2907769195820082,
"grad_norm": 0.77734375,
"learning_rate": 4.137151834863213e-05,
"loss": 0.9492,
"step": 160
},
{
"epoch": 0.3271240345297592,
"grad_norm": 0.7265625,
"learning_rate": 3.9062929058018275e-05,
"loss": 0.945,
"step": 180
},
{
"epoch": 0.36347114947751025,
"grad_norm": 0.76953125,
"learning_rate": 3.656345725602089e-05,
"loss": 0.93,
"step": 200
},
{
"epoch": 0.39981826442526125,
"grad_norm": 0.7421875,
"learning_rate": 3.390702940651737e-05,
"loss": 0.9214,
"step": 220
},
{
"epoch": 0.43616537937301225,
"grad_norm": 0.6953125,
"learning_rate": 3.1129702408972315e-05,
"loss": 0.9016,
"step": 240
},
{
"epoch": 0.4725124943207633,
"grad_norm": 0.6875,
"learning_rate": 2.8269174181795e-05,
"loss": 0.9047,
"step": 260
},
{
"epoch": 0.5088596092685143,
"grad_norm": 0.6953125,
"learning_rate": 2.536427197140289e-05,
"loss": 0.9123,
"step": 280
},
{
"epoch": 0.5452067242162654,
"grad_norm": 0.66796875,
"learning_rate": 2.2454425332404122e-05,
"loss": 0.8958,
"step": 300
},
{
"epoch": 0.5815538391640164,
"grad_norm": 0.6640625,
"learning_rate": 1.9579130932377774e-05,
"loss": 0.9014,
"step": 320
},
{
"epoch": 0.6179009541117674,
"grad_norm": 0.63671875,
"learning_rate": 1.6777416445699508e-05,
"loss": 0.9097,
"step": 340
},
{
"epoch": 0.6542480690595184,
"grad_norm": 0.59765625,
"learning_rate": 1.4087310813224414e-05,
"loss": 0.8954,
"step": 360
},
{
"epoch": 0.6905951840072694,
"grad_norm": 0.66015625,
"learning_rate": 1.15453280582328e-05,
"loss": 0.8896,
"step": 380
},
{
"epoch": 0.7269422989550205,
"grad_norm": 0.6328125,
"learning_rate": 9.185971665038959e-06,
"loss": 0.8946,
"step": 400
},
{
"epoch": 0.7632894139027715,
"grad_norm": 0.609375,
"learning_rate": 7.041266247556813e-06,
"loss": 0.8794,
"step": 420
},
{
"epoch": 0.7996365288505225,
"grad_norm": 0.65234375,
"learning_rate": 5.140322864697183e-06,
"loss": 0.8875,
"step": 440
},
{
"epoch": 0.8359836437982735,
"grad_norm": 0.62109375,
"learning_rate": 3.508943882768065e-06,
"loss": 0.8903,
"step": 460
},
{
"epoch": 0.8723307587460245,
"grad_norm": 0.66796875,
"learning_rate": 2.169272748259454e-06,
"loss": 0.9018,
"step": 480
},
{
"epoch": 0.9086778736937755,
"grad_norm": 1.453125,
"learning_rate": 1.1394934248056765e-06,
"loss": 0.8806,
"step": 500
},
{
"epoch": 0.9086778736937755,
"eval_loss": 0.9066952466964722,
"eval_runtime": 9.9793,
"eval_samples_per_second": 142.595,
"eval_steps_per_second": 4.509,
"step": 500
}
],
"logging_steps": 20,
"max_steps": 551,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 4.659803172699636e+18,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}