vitaleantonio's picture
Upload checkpoint DeepSeek-Coder-CONTROL-LEETCODE-1.3B-Base-2 at epoch 2
5b3d7fc verified
Raw
History Blame Contribute Delete
5.56 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 2.0,
"eval_steps": 500,
"global_step": 136,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.07421150278293136,
"grad_norm": 0.08251953125,
"learning_rate": 4.970588235294118e-05,
"loss": 0.7136603355407715,
"step": 5
},
{
"epoch": 0.14842300556586271,
"grad_norm": 0.07861328125,
"learning_rate": 4.933823529411765e-05,
"loss": 0.6834848880767822,
"step": 10
},
{
"epoch": 0.22263450834879406,
"grad_norm": 0.0830078125,
"learning_rate": 4.897058823529412e-05,
"loss": 0.707883644104004,
"step": 15
},
{
"epoch": 0.29684601113172543,
"grad_norm": 0.0859375,
"learning_rate": 4.860294117647059e-05,
"loss": 0.6775740623474121,
"step": 20
},
{
"epoch": 0.37105751391465674,
"grad_norm": 0.091796875,
"learning_rate": 4.823529411764706e-05,
"loss": 0.776744794845581,
"step": 25
},
{
"epoch": 0.4452690166975881,
"grad_norm": 0.0849609375,
"learning_rate": 4.7867647058823535e-05,
"loss": 0.728224515914917,
"step": 30
},
{
"epoch": 0.5194805194805194,
"grad_norm": 0.06884765625,
"learning_rate": 4.75e-05,
"loss": 0.6960197925567627,
"step": 35
},
{
"epoch": 0.5936920222634509,
"grad_norm": 0.08447265625,
"learning_rate": 4.713235294117647e-05,
"loss": 0.6611286163330078,
"step": 40
},
{
"epoch": 0.6679035250463822,
"grad_norm": 0.07666015625,
"learning_rate": 4.6764705882352944e-05,
"loss": 0.6967206954956054,
"step": 45
},
{
"epoch": 0.7421150278293135,
"grad_norm": 0.07470703125,
"learning_rate": 4.639705882352942e-05,
"loss": 0.6535059452056885,
"step": 50
},
{
"epoch": 0.8163265306122449,
"grad_norm": 0.0830078125,
"learning_rate": 4.6029411764705885e-05,
"loss": 0.70545973777771,
"step": 55
},
{
"epoch": 0.8905380333951762,
"grad_norm": 0.08740234375,
"learning_rate": 4.566176470588235e-05,
"loss": 0.6584103107452393,
"step": 60
},
{
"epoch": 0.9647495361781077,
"grad_norm": 0.0693359375,
"learning_rate": 4.5294117647058826e-05,
"loss": 0.6759594917297364,
"step": 65
},
{
"epoch": 1.0296846011131726,
"grad_norm": 0.0693359375,
"learning_rate": 4.49264705882353e-05,
"loss": 0.6613986492156982,
"step": 70
},
{
"epoch": 1.103896103896104,
"grad_norm": 0.061767578125,
"learning_rate": 4.455882352941177e-05,
"loss": 0.6283697128295899,
"step": 75
},
{
"epoch": 1.1781076066790352,
"grad_norm": 0.0751953125,
"learning_rate": 4.4191176470588235e-05,
"loss": 0.5669743061065674,
"step": 80
},
{
"epoch": 1.2523191094619666,
"grad_norm": 0.08544921875,
"learning_rate": 4.382352941176471e-05,
"loss": 0.5851097583770752,
"step": 85
},
{
"epoch": 1.3265306122448979,
"grad_norm": 0.076171875,
"learning_rate": 4.345588235294118e-05,
"loss": 0.59378023147583,
"step": 90
},
{
"epoch": 1.4007421150278292,
"grad_norm": 0.07080078125,
"learning_rate": 4.308823529411765e-05,
"loss": 0.561594820022583,
"step": 95
},
{
"epoch": 1.4749536178107607,
"grad_norm": 0.08447265625,
"learning_rate": 4.272058823529412e-05,
"loss": 0.5819839477539063,
"step": 100
},
{
"epoch": 1.549165120593692,
"grad_norm": 0.08056640625,
"learning_rate": 4.235294117647059e-05,
"loss": 0.5523958206176758,
"step": 105
},
{
"epoch": 1.6233766233766234,
"grad_norm": 0.06591796875,
"learning_rate": 4.198529411764706e-05,
"loss": 0.5537418365478516,
"step": 110
},
{
"epoch": 1.6975881261595547,
"grad_norm": 0.07470703125,
"learning_rate": 4.161764705882353e-05,
"loss": 0.551865530014038,
"step": 115
},
{
"epoch": 1.7717996289424862,
"grad_norm": 0.0703125,
"learning_rate": 4.125e-05,
"loss": 0.602507495880127,
"step": 120
},
{
"epoch": 1.8460111317254175,
"grad_norm": 0.0830078125,
"learning_rate": 4.0882352941176474e-05,
"loss": 0.5978893280029297,
"step": 125
},
{
"epoch": 1.9202226345083488,
"grad_norm": 0.08154296875,
"learning_rate": 4.051470588235294e-05,
"loss": 0.625958776473999,
"step": 130
},
{
"epoch": 1.9944341372912802,
"grad_norm": 0.0703125,
"learning_rate": 4.0147058823529415e-05,
"loss": 0.553464412689209,
"step": 135
}
],
"logging_steps": 5,
"max_steps": 680,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 6.778077240832819e+16,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}