vitaleantonio's picture
Upload checkpoint DeepSeek-Coder-CONTROL-MCEVALHARD-1.3B-Base-2 at epoch 2
8fbe5d7 verified
Raw
History Blame Contribute Delete
5.8 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 2.0,
"eval_steps": 500,
"global_step": 144,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.070298769771529,
"grad_norm": 0.041015625,
"learning_rate": 4.972222222222223e-05,
"loss": 0.6872987270355224,
"step": 5
},
{
"epoch": 0.140597539543058,
"grad_norm": 0.036376953125,
"learning_rate": 4.937500000000001e-05,
"loss": 0.6830456733703614,
"step": 10
},
{
"epoch": 0.210896309314587,
"grad_norm": 0.038330078125,
"learning_rate": 4.902777777777778e-05,
"loss": 0.7479233264923095,
"step": 15
},
{
"epoch": 0.281195079086116,
"grad_norm": 0.0380859375,
"learning_rate": 4.8680555555555554e-05,
"loss": 0.7160741329193115,
"step": 20
},
{
"epoch": 0.351493848857645,
"grad_norm": 0.03955078125,
"learning_rate": 4.8333333333333334e-05,
"loss": 0.6858582973480225,
"step": 25
},
{
"epoch": 0.421792618629174,
"grad_norm": 0.038818359375,
"learning_rate": 4.7986111111111113e-05,
"loss": 0.6775042533874511,
"step": 30
},
{
"epoch": 0.492091388400703,
"grad_norm": 0.03271484375,
"learning_rate": 4.7638888888888887e-05,
"loss": 0.6517538070678711,
"step": 35
},
{
"epoch": 0.562390158172232,
"grad_norm": 0.039306640625,
"learning_rate": 4.7291666666666666e-05,
"loss": 0.7560918807983399,
"step": 40
},
{
"epoch": 0.632688927943761,
"grad_norm": 0.037109375,
"learning_rate": 4.6944444444444446e-05,
"loss": 0.7019021511077881,
"step": 45
},
{
"epoch": 0.70298769771529,
"grad_norm": 0.03369140625,
"learning_rate": 4.6597222222222226e-05,
"loss": 0.6687473297119141,
"step": 50
},
{
"epoch": 0.773286467486819,
"grad_norm": 0.03369140625,
"learning_rate": 4.6250000000000006e-05,
"loss": 0.6842537879943847,
"step": 55
},
{
"epoch": 0.843585237258348,
"grad_norm": 0.03173828125,
"learning_rate": 4.590277777777778e-05,
"loss": 0.6241245269775391,
"step": 60
},
{
"epoch": 0.9138840070298769,
"grad_norm": 0.038818359375,
"learning_rate": 4.555555555555556e-05,
"loss": 0.6967287063598633,
"step": 65
},
{
"epoch": 0.984182776801406,
"grad_norm": 0.036376953125,
"learning_rate": 4.520833333333334e-05,
"loss": 0.6635632038116455,
"step": 70
},
{
"epoch": 1.0421792618629173,
"grad_norm": 0.032958984375,
"learning_rate": 4.486111111111111e-05,
"loss": 0.6175796985626221,
"step": 75
},
{
"epoch": 1.1124780316344465,
"grad_norm": 0.033203125,
"learning_rate": 4.4513888888888885e-05,
"loss": 0.5492826938629151,
"step": 80
},
{
"epoch": 1.1827768014059754,
"grad_norm": 0.047119140625,
"learning_rate": 4.4166666666666665e-05,
"loss": 0.5715642452239991,
"step": 85
},
{
"epoch": 1.2530755711775043,
"grad_norm": 0.03857421875,
"learning_rate": 4.3819444444444445e-05,
"loss": 0.6231389522552491,
"step": 90
},
{
"epoch": 1.3233743409490333,
"grad_norm": 0.040283203125,
"learning_rate": 4.3472222222222225e-05,
"loss": 0.5583981513977051,
"step": 95
},
{
"epoch": 1.3936731107205624,
"grad_norm": 0.04736328125,
"learning_rate": 4.3125000000000005e-05,
"loss": 0.5893191814422607,
"step": 100
},
{
"epoch": 1.4639718804920914,
"grad_norm": 0.041015625,
"learning_rate": 4.277777777777778e-05,
"loss": 0.5815530776977539,
"step": 105
},
{
"epoch": 1.5342706502636205,
"grad_norm": 0.0361328125,
"learning_rate": 4.243055555555556e-05,
"loss": 0.5973546504974365,
"step": 110
},
{
"epoch": 1.6045694200351495,
"grad_norm": 0.041259765625,
"learning_rate": 4.208333333333334e-05,
"loss": 0.5985394954681397,
"step": 115
},
{
"epoch": 1.6748681898066784,
"grad_norm": 0.0311279296875,
"learning_rate": 4.173611111111112e-05,
"loss": 0.5578798294067383,
"step": 120
},
{
"epoch": 1.7451669595782073,
"grad_norm": 0.036376953125,
"learning_rate": 4.138888888888889e-05,
"loss": 0.5978553295135498,
"step": 125
},
{
"epoch": 1.8154657293497363,
"grad_norm": 0.041748046875,
"learning_rate": 4.104166666666667e-05,
"loss": 0.6134780406951904,
"step": 130
},
{
"epoch": 1.8857644991212654,
"grad_norm": 0.041748046875,
"learning_rate": 4.0694444444444444e-05,
"loss": 0.5452652454376221,
"step": 135
},
{
"epoch": 1.9560632688927944,
"grad_norm": 0.03759765625,
"learning_rate": 4.0347222222222223e-05,
"loss": 0.5957761287689209,
"step": 140
}
],
"logging_steps": 5,
"max_steps": 720,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 7.15883274043392e+16,
"train_batch_size": 2,
"trial_name": null,
"trial_params": null
}