vitaleantonio's picture
Upload checkpoint DeepSeek-Coder-CONTROL-MCEVALHARD-6.7B-Base-3 at epoch 3
a3f8b46 verified
Raw
History Blame Contribute Delete
8.57 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 3.0,
"eval_steps": 500,
"global_step": 216,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.07032967032967033,
"grad_norm": 0.015869140625,
"learning_rate": 4.972222222222223e-05,
"loss": 0.6071998119354248,
"step": 5
},
{
"epoch": 0.14065934065934066,
"grad_norm": 0.01129150390625,
"learning_rate": 4.937500000000001e-05,
"loss": 0.6054113388061524,
"step": 10
},
{
"epoch": 0.210989010989011,
"grad_norm": 0.01287841796875,
"learning_rate": 4.902777777777778e-05,
"loss": 0.6566670417785645,
"step": 15
},
{
"epoch": 0.2813186813186813,
"grad_norm": 0.01220703125,
"learning_rate": 4.8680555555555554e-05,
"loss": 0.6298727989196777,
"step": 20
},
{
"epoch": 0.3516483516483517,
"grad_norm": 0.013916015625,
"learning_rate": 4.8333333333333334e-05,
"loss": 0.6060850143432617,
"step": 25
},
{
"epoch": 0.421978021978022,
"grad_norm": 0.01153564453125,
"learning_rate": 4.7986111111111113e-05,
"loss": 0.5983633995056152,
"step": 30
},
{
"epoch": 0.49230769230769234,
"grad_norm": 0.00970458984375,
"learning_rate": 4.7638888888888887e-05,
"loss": 0.5858489036560058,
"step": 35
},
{
"epoch": 0.5626373626373626,
"grad_norm": 0.01385498046875,
"learning_rate": 4.7291666666666666e-05,
"loss": 0.6647251129150391,
"step": 40
},
{
"epoch": 0.6329670329670329,
"grad_norm": 0.01177978515625,
"learning_rate": 4.6944444444444446e-05,
"loss": 0.606710147857666,
"step": 45
},
{
"epoch": 0.7032967032967034,
"grad_norm": 0.011962890625,
"learning_rate": 4.6597222222222226e-05,
"loss": 0.5918601512908935,
"step": 50
},
{
"epoch": 0.7736263736263737,
"grad_norm": 0.01129150390625,
"learning_rate": 4.6250000000000006e-05,
"loss": 0.6076582431793213,
"step": 55
},
{
"epoch": 0.843956043956044,
"grad_norm": 0.00958251953125,
"learning_rate": 4.590277777777778e-05,
"loss": 0.5553756713867187,
"step": 60
},
{
"epoch": 0.9142857142857143,
"grad_norm": 0.01348876953125,
"learning_rate": 4.555555555555556e-05,
"loss": 0.6189921379089356,
"step": 65
},
{
"epoch": 0.9846153846153847,
"grad_norm": 0.01220703125,
"learning_rate": 4.520833333333334e-05,
"loss": 0.5911207675933838,
"step": 70
},
{
"epoch": 1.0421978021978022,
"grad_norm": 0.00958251953125,
"learning_rate": 4.486111111111111e-05,
"loss": 0.5420764923095703,
"step": 75
},
{
"epoch": 1.1125274725274725,
"grad_norm": 0.01153564453125,
"learning_rate": 4.4513888888888885e-05,
"loss": 0.4824223518371582,
"step": 80
},
{
"epoch": 1.1828571428571428,
"grad_norm": 0.01434326171875,
"learning_rate": 4.4166666666666665e-05,
"loss": 0.5018988132476807,
"step": 85
},
{
"epoch": 1.2531868131868131,
"grad_norm": 0.013671875,
"learning_rate": 4.3819444444444445e-05,
"loss": 0.5336177825927735,
"step": 90
},
{
"epoch": 1.3235164835164834,
"grad_norm": 0.01263427734375,
"learning_rate": 4.3472222222222225e-05,
"loss": 0.4850775241851807,
"step": 95
},
{
"epoch": 1.393846153846154,
"grad_norm": 0.0126953125,
"learning_rate": 4.3125000000000005e-05,
"loss": 0.5138346195220947,
"step": 100
},
{
"epoch": 1.4641758241758243,
"grad_norm": 0.01397705078125,
"learning_rate": 4.277777777777778e-05,
"loss": 0.5070078849792481,
"step": 105
},
{
"epoch": 1.5345054945054946,
"grad_norm": 0.0118408203125,
"learning_rate": 4.243055555555556e-05,
"loss": 0.5198071479797364,
"step": 110
},
{
"epoch": 1.6048351648351649,
"grad_norm": 0.0137939453125,
"learning_rate": 4.208333333333334e-05,
"loss": 0.5197066783905029,
"step": 115
},
{
"epoch": 1.6751648351648352,
"grad_norm": 0.00970458984375,
"learning_rate": 4.173611111111112e-05,
"loss": 0.4952798366546631,
"step": 120
},
{
"epoch": 1.7454945054945055,
"grad_norm": 0.01190185546875,
"learning_rate": 4.138888888888889e-05,
"loss": 0.5169046878814697,
"step": 125
},
{
"epoch": 1.8158241758241758,
"grad_norm": 0.01324462890625,
"learning_rate": 4.104166666666667e-05,
"loss": 0.5308417320251465,
"step": 130
},
{
"epoch": 1.8861538461538463,
"grad_norm": 0.0146484375,
"learning_rate": 4.0694444444444444e-05,
"loss": 0.46563105583190917,
"step": 135
},
{
"epoch": 1.9564835164835164,
"grad_norm": 0.020751953125,
"learning_rate": 4.0347222222222223e-05,
"loss": 0.521865701675415,
"step": 140
},
{
"epoch": 2.014065934065934,
"grad_norm": 0.01312255859375,
"learning_rate": 4e-05,
"loss": 0.44051213264465333,
"step": 145
},
{
"epoch": 2.0843956043956045,
"grad_norm": 0.01263427734375,
"learning_rate": 3.9652777777777776e-05,
"loss": 0.43095035552978517,
"step": 150
},
{
"epoch": 2.1547252747252745,
"grad_norm": 0.020263671875,
"learning_rate": 3.9305555555555556e-05,
"loss": 0.4152026653289795,
"step": 155
},
{
"epoch": 2.225054945054945,
"grad_norm": 0.0120849609375,
"learning_rate": 3.8958333333333336e-05,
"loss": 0.43182148933410647,
"step": 160
},
{
"epoch": 2.2953846153846156,
"grad_norm": 0.0162353515625,
"learning_rate": 3.8611111111111116e-05,
"loss": 0.4308079719543457,
"step": 165
},
{
"epoch": 2.3657142857142857,
"grad_norm": 0.01300048828125,
"learning_rate": 3.826388888888889e-05,
"loss": 0.40520739555358887,
"step": 170
},
{
"epoch": 2.436043956043956,
"grad_norm": 0.015380859375,
"learning_rate": 3.791666666666667e-05,
"loss": 0.4348421096801758,
"step": 175
},
{
"epoch": 2.5063736263736263,
"grad_norm": 0.0166015625,
"learning_rate": 3.756944444444445e-05,
"loss": 0.3993671894073486,
"step": 180
},
{
"epoch": 2.576703296703297,
"grad_norm": 0.01611328125,
"learning_rate": 3.722222222222222e-05,
"loss": 0.456925630569458,
"step": 185
},
{
"epoch": 2.647032967032967,
"grad_norm": 0.0206298828125,
"learning_rate": 3.6875e-05,
"loss": 0.4396049976348877,
"step": 190
},
{
"epoch": 2.7173626373626374,
"grad_norm": 0.014404296875,
"learning_rate": 3.6527777777777775e-05,
"loss": 0.41503162384033204,
"step": 195
},
{
"epoch": 2.787692307692308,
"grad_norm": 0.01446533203125,
"learning_rate": 3.6180555555555555e-05,
"loss": 0.36976258754730223,
"step": 200
},
{
"epoch": 2.858021978021978,
"grad_norm": 0.0166015625,
"learning_rate": 3.5833333333333335e-05,
"loss": 0.4395163059234619,
"step": 205
},
{
"epoch": 2.9283516483516485,
"grad_norm": 0.0162353515625,
"learning_rate": 3.5486111111111115e-05,
"loss": 0.4275169849395752,
"step": 210
},
{
"epoch": 2.9986813186813186,
"grad_norm": 0.01544189453125,
"learning_rate": 3.513888888888889e-05,
"loss": 0.42092361450195315,
"step": 215
}
],
"logging_steps": 5,
"max_steps": 720,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 5.542167762173952e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}