vitaleantonio's picture
Upload checkpoint Qwen2.5-Coder-CONTROL-LEETCODE-7B-Base-2 at epoch 2
caff0b3 verified
Raw
History Blame Contribute Delete
5.59 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 2.0,
"eval_steps": 500,
"global_step": 136,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.07441860465116279,
"grad_norm": 0.04150390625,
"learning_rate": 4.970588235294118e-05,
"loss": 0.8117074966430664,
"step": 5
},
{
"epoch": 0.14883720930232558,
"grad_norm": 0.052001953125,
"learning_rate": 4.933823529411765e-05,
"loss": 0.8412944793701171,
"step": 10
},
{
"epoch": 0.22325581395348837,
"grad_norm": 0.038818359375,
"learning_rate": 4.897058823529412e-05,
"loss": 0.783283805847168,
"step": 15
},
{
"epoch": 0.29767441860465116,
"grad_norm": 0.043212890625,
"learning_rate": 4.860294117647059e-05,
"loss": 0.8191794395446778,
"step": 20
},
{
"epoch": 0.37209302325581395,
"grad_norm": 0.0390625,
"learning_rate": 4.823529411764706e-05,
"loss": 0.777860689163208,
"step": 25
},
{
"epoch": 0.44651162790697674,
"grad_norm": 0.04541015625,
"learning_rate": 4.7867647058823535e-05,
"loss": 0.8589226722717285,
"step": 30
},
{
"epoch": 0.5209302325581395,
"grad_norm": 0.041015625,
"learning_rate": 4.75e-05,
"loss": 0.8166523933410644,
"step": 35
},
{
"epoch": 0.5953488372093023,
"grad_norm": 0.0400390625,
"learning_rate": 4.713235294117647e-05,
"loss": 0.8092836380004883,
"step": 40
},
{
"epoch": 0.6697674418604651,
"grad_norm": 0.042724609375,
"learning_rate": 4.6764705882352944e-05,
"loss": 0.8805876731872558,
"step": 45
},
{
"epoch": 0.7441860465116279,
"grad_norm": 0.03955078125,
"learning_rate": 4.639705882352942e-05,
"loss": 0.838012409210205,
"step": 50
},
{
"epoch": 0.8186046511627907,
"grad_norm": 0.0419921875,
"learning_rate": 4.6029411764705885e-05,
"loss": 0.8522647857666016,
"step": 55
},
{
"epoch": 0.8930232558139535,
"grad_norm": 0.039306640625,
"learning_rate": 4.566176470588235e-05,
"loss": 0.7979763507843017,
"step": 60
},
{
"epoch": 0.9674418604651163,
"grad_norm": 0.037353515625,
"learning_rate": 4.5294117647058826e-05,
"loss": 0.7971380233764649,
"step": 65
},
{
"epoch": 1.0297674418604652,
"grad_norm": 0.033935546875,
"learning_rate": 4.49264705882353e-05,
"loss": 0.7957266807556153,
"step": 70
},
{
"epoch": 1.104186046511628,
"grad_norm": 0.032958984375,
"learning_rate": 4.455882352941177e-05,
"loss": 0.5031315326690674,
"step": 75
},
{
"epoch": 1.1786046511627908,
"grad_norm": 0.0517578125,
"learning_rate": 4.4191176470588235e-05,
"loss": 0.5382503509521485,
"step": 80
},
{
"epoch": 1.2530232558139534,
"grad_norm": 0.034912109375,
"learning_rate": 4.382352941176471e-05,
"loss": 0.5200568199157715,
"step": 85
},
{
"epoch": 1.3274418604651164,
"grad_norm": 0.033935546875,
"learning_rate": 4.345588235294118e-05,
"loss": 0.5138055801391601,
"step": 90
},
{
"epoch": 1.4018604651162792,
"grad_norm": 0.040771484375,
"learning_rate": 4.308823529411765e-05,
"loss": 0.4661245822906494,
"step": 95
},
{
"epoch": 1.476279069767442,
"grad_norm": 0.04248046875,
"learning_rate": 4.272058823529412e-05,
"loss": 0.5133561611175537,
"step": 100
},
{
"epoch": 1.5506976744186045,
"grad_norm": 0.031982421875,
"learning_rate": 4.235294117647059e-05,
"loss": 0.4750948905944824,
"step": 105
},
{
"epoch": 1.6251162790697675,
"grad_norm": 0.031494140625,
"learning_rate": 4.198529411764706e-05,
"loss": 0.46523518562316896,
"step": 110
},
{
"epoch": 1.69953488372093,
"grad_norm": 0.03515625,
"learning_rate": 4.161764705882353e-05,
"loss": 0.4725958824157715,
"step": 115
},
{
"epoch": 1.773953488372093,
"grad_norm": 0.0439453125,
"learning_rate": 4.125e-05,
"loss": 0.4394645690917969,
"step": 120
},
{
"epoch": 1.8483720930232557,
"grad_norm": 0.03515625,
"learning_rate": 4.0882352941176474e-05,
"loss": 0.463104772567749,
"step": 125
},
{
"epoch": 1.9227906976744187,
"grad_norm": 0.03955078125,
"learning_rate": 4.051470588235294e-05,
"loss": 0.5168607711791993,
"step": 130
},
{
"epoch": 1.9972093023255812,
"grad_norm": 0.04638671875,
"learning_rate": 4.0147058823529415e-05,
"loss": 0.5063523292541504,
"step": 135
}
],
"logging_steps": 5,
"max_steps": 680,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 3.736002021556224e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}