mo14_sft_adapters / full_transcript /03_MED /trainer_state.json
jprivera44's picture
Add full_transcript/03_MED final adapter (checkpoint-215)
c0e4432 verified
Raw
History Blame Contribute Delete
7.51 kB
{
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 215,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.023255813953488372,
"grad_norm": 0.3701596260070801,
"learning_rate": 2e-05,
"loss": 1.2482,
"step": 5
},
{
"epoch": 0.046511627906976744,
"grad_norm": 0.22348932921886444,
"learning_rate": 2e-05,
"loss": 1.0639,
"step": 10
},
{
"epoch": 0.06976744186046512,
"grad_norm": 0.30826476216316223,
"learning_rate": 2e-05,
"loss": 0.975,
"step": 15
},
{
"epoch": 0.09302325581395349,
"grad_norm": 0.19094961881637573,
"learning_rate": 2e-05,
"loss": 1.0374,
"step": 20
},
{
"epoch": 0.11627906976744186,
"grad_norm": 0.18843892216682434,
"learning_rate": 2e-05,
"loss": 0.9062,
"step": 25
},
{
"epoch": 0.13953488372093023,
"grad_norm": 0.15660837292671204,
"learning_rate": 2e-05,
"loss": 0.8813,
"step": 30
},
{
"epoch": 0.16279069767441862,
"grad_norm": 0.2319907695055008,
"learning_rate": 2e-05,
"loss": 0.9589,
"step": 35
},
{
"epoch": 0.18604651162790697,
"grad_norm": 0.20357345044612885,
"learning_rate": 2e-05,
"loss": 0.9257,
"step": 40
},
{
"epoch": 0.20930232558139536,
"grad_norm": 0.16901366412639618,
"learning_rate": 2e-05,
"loss": 0.8194,
"step": 45
},
{
"epoch": 0.23255813953488372,
"grad_norm": 0.24122780561447144,
"learning_rate": 2e-05,
"loss": 0.8726,
"step": 50
},
{
"epoch": 0.2558139534883721,
"grad_norm": 0.21689686179161072,
"learning_rate": 2e-05,
"loss": 0.9835,
"step": 55
},
{
"epoch": 0.27906976744186046,
"grad_norm": 0.18955062329769135,
"learning_rate": 2e-05,
"loss": 0.8624,
"step": 60
},
{
"epoch": 0.3023255813953488,
"grad_norm": 0.2065642923116684,
"learning_rate": 2e-05,
"loss": 0.8287,
"step": 65
},
{
"epoch": 0.32558139534883723,
"grad_norm": 0.18894456326961517,
"learning_rate": 2e-05,
"loss": 0.981,
"step": 70
},
{
"epoch": 0.3488372093023256,
"grad_norm": 0.17258857190608978,
"learning_rate": 2e-05,
"loss": 0.8652,
"step": 75
},
{
"epoch": 0.37209302325581395,
"grad_norm": 0.1549784541130066,
"learning_rate": 2e-05,
"loss": 0.8775,
"step": 80
},
{
"epoch": 0.3953488372093023,
"grad_norm": 0.18814615905284882,
"learning_rate": 2e-05,
"loss": 0.9429,
"step": 85
},
{
"epoch": 0.4186046511627907,
"grad_norm": 0.18734842538833618,
"learning_rate": 2e-05,
"loss": 0.8884,
"step": 90
},
{
"epoch": 0.4418604651162791,
"grad_norm": 0.1647917777299881,
"learning_rate": 2e-05,
"loss": 0.8504,
"step": 95
},
{
"epoch": 0.46511627906976744,
"grad_norm": 0.24890270829200745,
"learning_rate": 2e-05,
"loss": 0.8224,
"step": 100
},
{
"epoch": 0.4883720930232558,
"grad_norm": 0.20458489656448364,
"learning_rate": 2e-05,
"loss": 0.9412,
"step": 105
},
{
"epoch": 0.5116279069767442,
"grad_norm": 0.18102170526981354,
"learning_rate": 2e-05,
"loss": 0.8754,
"step": 110
},
{
"epoch": 0.5348837209302325,
"grad_norm": 0.17741850018501282,
"learning_rate": 2e-05,
"loss": 0.8302,
"step": 115
},
{
"epoch": 0.5581395348837209,
"grad_norm": 0.20435400307178497,
"learning_rate": 2e-05,
"loss": 0.9222,
"step": 120
},
{
"epoch": 0.5813953488372093,
"grad_norm": 0.19005118310451508,
"learning_rate": 2e-05,
"loss": 0.8278,
"step": 125
},
{
"epoch": 0.6046511627906976,
"grad_norm": 0.17868107557296753,
"learning_rate": 2e-05,
"loss": 0.8354,
"step": 130
},
{
"epoch": 0.627906976744186,
"grad_norm": 0.1938895732164383,
"learning_rate": 2e-05,
"loss": 0.902,
"step": 135
},
{
"epoch": 0.6511627906976745,
"grad_norm": 0.18117573857307434,
"learning_rate": 2e-05,
"loss": 0.8911,
"step": 140
},
{
"epoch": 0.6744186046511628,
"grad_norm": 0.21764567494392395,
"learning_rate": 2e-05,
"loss": 0.8493,
"step": 145
},
{
"epoch": 0.6976744186046512,
"grad_norm": 0.3313971161842346,
"learning_rate": 2e-05,
"loss": 0.8318,
"step": 150
},
{
"epoch": 0.7209302325581395,
"grad_norm": 0.22705790400505066,
"learning_rate": 2e-05,
"loss": 0.9222,
"step": 155
},
{
"epoch": 0.7441860465116279,
"grad_norm": 0.183769091963768,
"learning_rate": 2e-05,
"loss": 0.8331,
"step": 160
},
{
"epoch": 0.7674418604651163,
"grad_norm": 0.17954127490520477,
"learning_rate": 2e-05,
"loss": 0.8094,
"step": 165
},
{
"epoch": 0.7906976744186046,
"grad_norm": 0.1975155919790268,
"learning_rate": 2e-05,
"loss": 0.8994,
"step": 170
},
{
"epoch": 0.813953488372093,
"grad_norm": 0.21705932915210724,
"learning_rate": 2e-05,
"loss": 0.797,
"step": 175
},
{
"epoch": 0.8372093023255814,
"grad_norm": 0.18831680715084076,
"learning_rate": 2e-05,
"loss": 0.8198,
"step": 180
},
{
"epoch": 0.8604651162790697,
"grad_norm": 0.23201768100261688,
"learning_rate": 2e-05,
"loss": 0.8965,
"step": 185
},
{
"epoch": 0.8837209302325582,
"grad_norm": 0.20603500306606293,
"learning_rate": 2e-05,
"loss": 0.8632,
"step": 190
},
{
"epoch": 0.9069767441860465,
"grad_norm": 0.19485333561897278,
"learning_rate": 2e-05,
"loss": 0.8264,
"step": 195
},
{
"epoch": 0.9302325581395349,
"grad_norm": 0.23732714354991913,
"learning_rate": 2e-05,
"loss": 0.8268,
"step": 200
},
{
"epoch": 0.9534883720930233,
"grad_norm": 0.22331510484218597,
"learning_rate": 2e-05,
"loss": 0.9291,
"step": 205
},
{
"epoch": 0.9767441860465116,
"grad_norm": 0.18885022401809692,
"learning_rate": 2e-05,
"loss": 0.8126,
"step": 210
},
{
"epoch": 1.0,
"grad_norm": 0.19218967854976654,
"learning_rate": 2e-05,
"loss": 0.8162,
"step": 215
}
],
"logging_steps": 5,
"max_steps": 215,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 99999,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 3.1971222282777395e+17,
"train_batch_size": 8,
"trial_name": null,
"trial_params": null
}