vitaleantonio's picture
Upload checkpoint DeepSeek-Coder-CONTROL-LEETCODE-1.3B-Base-3 at epoch 3
578f8e0 verified
Raw
History Blame Contribute Delete
7.9 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 3.0,
"eval_steps": 500,
"global_step": 204,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.07421150278293136,
"grad_norm": 0.08251953125,
"learning_rate": 4.970588235294118e-05,
"loss": 0.7136603355407715,
"step": 5
},
{
"epoch": 0.14842300556586271,
"grad_norm": 0.07861328125,
"learning_rate": 4.933823529411765e-05,
"loss": 0.6834848880767822,
"step": 10
},
{
"epoch": 0.22263450834879406,
"grad_norm": 0.0830078125,
"learning_rate": 4.897058823529412e-05,
"loss": 0.707883644104004,
"step": 15
},
{
"epoch": 0.29684601113172543,
"grad_norm": 0.0859375,
"learning_rate": 4.860294117647059e-05,
"loss": 0.6775740623474121,
"step": 20
},
{
"epoch": 0.37105751391465674,
"grad_norm": 0.091796875,
"learning_rate": 4.823529411764706e-05,
"loss": 0.776744794845581,
"step": 25
},
{
"epoch": 0.4452690166975881,
"grad_norm": 0.0849609375,
"learning_rate": 4.7867647058823535e-05,
"loss": 0.728224515914917,
"step": 30
},
{
"epoch": 0.5194805194805194,
"grad_norm": 0.06884765625,
"learning_rate": 4.75e-05,
"loss": 0.6960197925567627,
"step": 35
},
{
"epoch": 0.5936920222634509,
"grad_norm": 0.08447265625,
"learning_rate": 4.713235294117647e-05,
"loss": 0.6611286163330078,
"step": 40
},
{
"epoch": 0.6679035250463822,
"grad_norm": 0.07666015625,
"learning_rate": 4.6764705882352944e-05,
"loss": 0.6967206954956054,
"step": 45
},
{
"epoch": 0.7421150278293135,
"grad_norm": 0.07470703125,
"learning_rate": 4.639705882352942e-05,
"loss": 0.6535059452056885,
"step": 50
},
{
"epoch": 0.8163265306122449,
"grad_norm": 0.0830078125,
"learning_rate": 4.6029411764705885e-05,
"loss": 0.70545973777771,
"step": 55
},
{
"epoch": 0.8905380333951762,
"grad_norm": 0.08740234375,
"learning_rate": 4.566176470588235e-05,
"loss": 0.6584103107452393,
"step": 60
},
{
"epoch": 0.9647495361781077,
"grad_norm": 0.0693359375,
"learning_rate": 4.5294117647058826e-05,
"loss": 0.6759594917297364,
"step": 65
},
{
"epoch": 1.0296846011131726,
"grad_norm": 0.0693359375,
"learning_rate": 4.49264705882353e-05,
"loss": 0.6613986492156982,
"step": 70
},
{
"epoch": 1.103896103896104,
"grad_norm": 0.061767578125,
"learning_rate": 4.455882352941177e-05,
"loss": 0.6283697128295899,
"step": 75
},
{
"epoch": 1.1781076066790352,
"grad_norm": 0.0751953125,
"learning_rate": 4.4191176470588235e-05,
"loss": 0.5669743061065674,
"step": 80
},
{
"epoch": 1.2523191094619666,
"grad_norm": 0.08544921875,
"learning_rate": 4.382352941176471e-05,
"loss": 0.5851097583770752,
"step": 85
},
{
"epoch": 1.3265306122448979,
"grad_norm": 0.076171875,
"learning_rate": 4.345588235294118e-05,
"loss": 0.59378023147583,
"step": 90
},
{
"epoch": 1.4007421150278292,
"grad_norm": 0.07080078125,
"learning_rate": 4.308823529411765e-05,
"loss": 0.561594820022583,
"step": 95
},
{
"epoch": 1.4749536178107607,
"grad_norm": 0.08447265625,
"learning_rate": 4.272058823529412e-05,
"loss": 0.5819839477539063,
"step": 100
},
{
"epoch": 1.549165120593692,
"grad_norm": 0.08056640625,
"learning_rate": 4.235294117647059e-05,
"loss": 0.5523958206176758,
"step": 105
},
{
"epoch": 1.6233766233766234,
"grad_norm": 0.06591796875,
"learning_rate": 4.198529411764706e-05,
"loss": 0.5537418365478516,
"step": 110
},
{
"epoch": 1.6975881261595547,
"grad_norm": 0.07470703125,
"learning_rate": 4.161764705882353e-05,
"loss": 0.551865530014038,
"step": 115
},
{
"epoch": 1.7717996289424862,
"grad_norm": 0.0703125,
"learning_rate": 4.125e-05,
"loss": 0.602507495880127,
"step": 120
},
{
"epoch": 1.8460111317254175,
"grad_norm": 0.0830078125,
"learning_rate": 4.0882352941176474e-05,
"loss": 0.5978893280029297,
"step": 125
},
{
"epoch": 1.9202226345083488,
"grad_norm": 0.08154296875,
"learning_rate": 4.051470588235294e-05,
"loss": 0.625958776473999,
"step": 130
},
{
"epoch": 1.9944341372912802,
"grad_norm": 0.0703125,
"learning_rate": 4.0147058823529415e-05,
"loss": 0.553464412689209,
"step": 135
},
{
"epoch": 2.0593692022263452,
"grad_norm": 0.07568359375,
"learning_rate": 3.977941176470588e-05,
"loss": 0.5257774353027344,
"step": 140
},
{
"epoch": 2.1335807050092765,
"grad_norm": 0.07666015625,
"learning_rate": 3.9411764705882356e-05,
"loss": 0.5616061687469482,
"step": 145
},
{
"epoch": 2.207792207792208,
"grad_norm": 0.08447265625,
"learning_rate": 3.9044117647058823e-05,
"loss": 0.486341667175293,
"step": 150
},
{
"epoch": 2.282003710575139,
"grad_norm": 0.0830078125,
"learning_rate": 3.86764705882353e-05,
"loss": 0.4711045265197754,
"step": 155
},
{
"epoch": 2.3562152133580705,
"grad_norm": 0.091796875,
"learning_rate": 3.830882352941177e-05,
"loss": 0.47898192405700685,
"step": 160
},
{
"epoch": 2.430426716141002,
"grad_norm": 0.07861328125,
"learning_rate": 3.794117647058824e-05,
"loss": 0.49690723419189453,
"step": 165
},
{
"epoch": 2.504638218923933,
"grad_norm": 0.08544921875,
"learning_rate": 3.7573529411764706e-05,
"loss": 0.5519385814666748,
"step": 170
},
{
"epoch": 2.5788497217068644,
"grad_norm": 0.08642578125,
"learning_rate": 3.720588235294118e-05,
"loss": 0.5011870384216308,
"step": 175
},
{
"epoch": 2.6530612244897958,
"grad_norm": 0.0908203125,
"learning_rate": 3.6838235294117654e-05,
"loss": 0.5261661052703858,
"step": 180
},
{
"epoch": 2.7272727272727275,
"grad_norm": 0.07373046875,
"learning_rate": 3.6470588235294114e-05,
"loss": 0.49692234992980955,
"step": 185
},
{
"epoch": 2.8014842300556584,
"grad_norm": 0.0859375,
"learning_rate": 3.610294117647059e-05,
"loss": 0.48595056533813474,
"step": 190
},
{
"epoch": 2.87569573283859,
"grad_norm": 0.07177734375,
"learning_rate": 3.573529411764706e-05,
"loss": 0.46908016204833985,
"step": 195
},
{
"epoch": 2.9499072356215215,
"grad_norm": 0.0947265625,
"learning_rate": 3.5367647058823536e-05,
"loss": 0.5229842662811279,
"step": 200
}
],
"logging_steps": 5,
"max_steps": 680,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 1.0167115861249229e+17,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}