vitaleantonio's picture
Upload checkpoint DeepSeek-Coder-CONTROL-LEETCODE-6.7B-Base-3 at epoch 3
25870df verified
Raw
History Blame Contribute Delete
8.02 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 3.0,
"eval_steps": 500,
"global_step": 204,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.07428040854224698,
"grad_norm": 0.0166015625,
"learning_rate": 4.970588235294118e-05,
"loss": 0.6377665996551514,
"step": 5
},
{
"epoch": 0.14856081708449395,
"grad_norm": 0.0146484375,
"learning_rate": 4.933823529411765e-05,
"loss": 0.6021791934967041,
"step": 10
},
{
"epoch": 0.22284122562674094,
"grad_norm": 0.01373291015625,
"learning_rate": 4.897058823529412e-05,
"loss": 0.6274982929229737,
"step": 15
},
{
"epoch": 0.2971216341689879,
"grad_norm": 0.0135498046875,
"learning_rate": 4.860294117647059e-05,
"loss": 0.599825382232666,
"step": 20
},
{
"epoch": 0.3714020427112349,
"grad_norm": 0.0159912109375,
"learning_rate": 4.823529411764706e-05,
"loss": 0.6859479427337647,
"step": 25
},
{
"epoch": 0.4456824512534819,
"grad_norm": 0.01422119140625,
"learning_rate": 4.7867647058823535e-05,
"loss": 0.636006212234497,
"step": 30
},
{
"epoch": 0.5199628597957289,
"grad_norm": 0.01116943359375,
"learning_rate": 4.75e-05,
"loss": 0.6127719879150391,
"step": 35
},
{
"epoch": 0.5942432683379758,
"grad_norm": 0.01531982421875,
"learning_rate": 4.713235294117647e-05,
"loss": 0.5927554130554199,
"step": 40
},
{
"epoch": 0.6685236768802229,
"grad_norm": 0.011474609375,
"learning_rate": 4.6764705882352944e-05,
"loss": 0.6282303333282471,
"step": 45
},
{
"epoch": 0.7428040854224698,
"grad_norm": 0.015869140625,
"learning_rate": 4.639705882352942e-05,
"loss": 0.5779662609100342,
"step": 50
},
{
"epoch": 0.8170844939647168,
"grad_norm": 0.01318359375,
"learning_rate": 4.6029411764705885e-05,
"loss": 0.617146348953247,
"step": 55
},
{
"epoch": 0.8913649025069638,
"grad_norm": 0.0130615234375,
"learning_rate": 4.566176470588235e-05,
"loss": 0.5746903896331788,
"step": 60
},
{
"epoch": 0.9656453110492108,
"grad_norm": 0.0108642578125,
"learning_rate": 4.5294117647058826e-05,
"loss": 0.5909358024597168,
"step": 65
},
{
"epoch": 1.0297121634168989,
"grad_norm": 0.00994873046875,
"learning_rate": 4.49264705882353e-05,
"loss": 0.5567221164703369,
"step": 70
},
{
"epoch": 1.1039925719591457,
"grad_norm": 0.01031494140625,
"learning_rate": 4.455882352941177e-05,
"loss": 0.5583184242248536,
"step": 75
},
{
"epoch": 1.1782729805013927,
"grad_norm": 0.01214599609375,
"learning_rate": 4.4191176470588235e-05,
"loss": 0.5047792911529541,
"step": 80
},
{
"epoch": 1.2525533890436398,
"grad_norm": 0.014892578125,
"learning_rate": 4.382352941176471e-05,
"loss": 0.509721040725708,
"step": 85
},
{
"epoch": 1.3268337975858868,
"grad_norm": 0.0162353515625,
"learning_rate": 4.345588235294118e-05,
"loss": 0.522771167755127,
"step": 90
},
{
"epoch": 1.4011142061281336,
"grad_norm": 0.01263427734375,
"learning_rate": 4.308823529411765e-05,
"loss": 0.48066258430480957,
"step": 95
},
{
"epoch": 1.4753946146703807,
"grad_norm": 0.02294921875,
"learning_rate": 4.272058823529412e-05,
"loss": 0.5085556030273437,
"step": 100
},
{
"epoch": 1.5496750232126275,
"grad_norm": 0.01287841796875,
"learning_rate": 4.235294117647059e-05,
"loss": 0.4806520462036133,
"step": 105
},
{
"epoch": 1.6239554317548746,
"grad_norm": 0.0107421875,
"learning_rate": 4.198529411764706e-05,
"loss": 0.48960013389587403,
"step": 110
},
{
"epoch": 1.6982358402971216,
"grad_norm": 0.0120849609375,
"learning_rate": 4.161764705882353e-05,
"loss": 0.47647743225097655,
"step": 115
},
{
"epoch": 1.7725162488393686,
"grad_norm": 0.0230712890625,
"learning_rate": 4.125e-05,
"loss": 0.5240037918090821,
"step": 120
},
{
"epoch": 1.8467966573816157,
"grad_norm": 0.0152587890625,
"learning_rate": 4.0882352941176474e-05,
"loss": 0.5106610298156739,
"step": 125
},
{
"epoch": 1.9210770659238627,
"grad_norm": 0.01312255859375,
"learning_rate": 4.051470588235294e-05,
"loss": 0.5423622608184815,
"step": 130
},
{
"epoch": 1.9953574744661096,
"grad_norm": 0.01214599609375,
"learning_rate": 4.0147058823529415e-05,
"loss": 0.48050861358642577,
"step": 135
},
{
"epoch": 2.0594243268337977,
"grad_norm": 0.01177978515625,
"learning_rate": 3.977941176470588e-05,
"loss": 0.4528938293457031,
"step": 140
},
{
"epoch": 2.1337047353760448,
"grad_norm": 0.0135498046875,
"learning_rate": 3.9411764705882356e-05,
"loss": 0.4708698749542236,
"step": 145
},
{
"epoch": 2.2079851439182914,
"grad_norm": 0.015869140625,
"learning_rate": 3.9044117647058823e-05,
"loss": 0.40610790252685547,
"step": 150
},
{
"epoch": 2.2822655524605384,
"grad_norm": 0.020263671875,
"learning_rate": 3.86764705882353e-05,
"loss": 0.4004369735717773,
"step": 155
},
{
"epoch": 2.3565459610027855,
"grad_norm": 0.01495361328125,
"learning_rate": 3.830882352941177e-05,
"loss": 0.402101993560791,
"step": 160
},
{
"epoch": 2.4308263695450325,
"grad_norm": 0.01519775390625,
"learning_rate": 3.794117647058824e-05,
"loss": 0.41231889724731446,
"step": 165
},
{
"epoch": 2.5051067780872796,
"grad_norm": 0.01611328125,
"learning_rate": 3.7573529411764706e-05,
"loss": 0.4784365653991699,
"step": 170
},
{
"epoch": 2.5793871866295266,
"grad_norm": 0.0174560546875,
"learning_rate": 3.720588235294118e-05,
"loss": 0.4141225814819336,
"step": 175
},
{
"epoch": 2.6536675951717736,
"grad_norm": 0.0189208984375,
"learning_rate": 3.6838235294117654e-05,
"loss": 0.44509315490722656,
"step": 180
},
{
"epoch": 2.7279480037140207,
"grad_norm": 0.018310546875,
"learning_rate": 3.6470588235294114e-05,
"loss": 0.4210689067840576,
"step": 185
},
{
"epoch": 2.8022284122562673,
"grad_norm": 0.0272216796875,
"learning_rate": 3.610294117647059e-05,
"loss": 0.4066977024078369,
"step": 190
},
{
"epoch": 2.8765088207985143,
"grad_norm": 0.0128173828125,
"learning_rate": 3.573529411764706e-05,
"loss": 0.4026792049407959,
"step": 195
},
{
"epoch": 2.9507892293407614,
"grad_norm": 0.0174560546875,
"learning_rate": 3.5367647058823536e-05,
"loss": 0.44449539184570314,
"step": 200
}
],
"logging_steps": 5,
"max_steps": 680,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 5.2473975207572275e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}