8b_v2_1epoch / trainer_state.json
Leoxx's picture
Upload folder using huggingface_hub
105f492 verified
Raw
History Blame Contribute Delete
6.34 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0681362725450902,
"eval_steps": 100.0,
"global_step": 800,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.033400133600534405,
"grad_norm": 34.83279037475586,
"learning_rate": 1.978638184245661e-05,
"loss": 5.0239,
"step": 25
},
{
"epoch": 0.06680026720106881,
"grad_norm": 21.53448486328125,
"learning_rate": 1.956386292834891e-05,
"loss": 4.5061,
"step": 50
},
{
"epoch": 0.10020040080160321,
"grad_norm": 23.527015686035156,
"learning_rate": 1.9341344014241213e-05,
"loss": 4.3217,
"step": 75
},
{
"epoch": 0.13360053440213762,
"grad_norm": 15.6005859375,
"learning_rate": 1.9118825100133513e-05,
"loss": 4.0712,
"step": 100
},
{
"epoch": 0.16700066800267202,
"grad_norm": 19.26547622680664,
"learning_rate": 1.8896306186025813e-05,
"loss": 4.0589,
"step": 125
},
{
"epoch": 0.20040080160320642,
"grad_norm": 23.404224395751953,
"learning_rate": 1.8673787271918113e-05,
"loss": 3.9447,
"step": 150
},
{
"epoch": 0.23380093520374082,
"grad_norm": 19.686691284179688,
"learning_rate": 1.8451268357810413e-05,
"loss": 3.6344,
"step": 175
},
{
"epoch": 0.26720106880427524,
"grad_norm": 20.32077407836914,
"learning_rate": 1.8228749443702717e-05,
"loss": 3.489,
"step": 200
},
{
"epoch": 0.30060120240480964,
"grad_norm": 17.107595443725586,
"learning_rate": 1.8006230529595017e-05,
"loss": 3.4025,
"step": 225
},
{
"epoch": 0.33400133600534404,
"grad_norm": 17.202909469604492,
"learning_rate": 1.7783711615487317e-05,
"loss": 3.1437,
"step": 250
},
{
"epoch": 0.36740146960587844,
"grad_norm": 21.202510833740234,
"learning_rate": 1.756119270137962e-05,
"loss": 3.2257,
"step": 275
},
{
"epoch": 0.40080160320641284,
"grad_norm": 19.31769371032715,
"learning_rate": 1.733867378727192e-05,
"loss": 3.1046,
"step": 300
},
{
"epoch": 0.43420173680694724,
"grad_norm": 21.412029266357422,
"learning_rate": 1.711615487316422e-05,
"loss": 3.2254,
"step": 325
},
{
"epoch": 0.46760187040748163,
"grad_norm": 23.345766067504883,
"learning_rate": 1.689363595905652e-05,
"loss": 3.1176,
"step": 350
},
{
"epoch": 0.501002004008016,
"grad_norm": 21.88970375061035,
"learning_rate": 1.667111704494882e-05,
"loss": 2.9884,
"step": 375
},
{
"epoch": 0.5344021376085505,
"grad_norm": 21.60796356201172,
"learning_rate": 1.6448598130841123e-05,
"loss": 3.1272,
"step": 400
},
{
"epoch": 0.5678022712090849,
"grad_norm": 17.43869972229004,
"learning_rate": 1.6226079216733423e-05,
"loss": 2.8604,
"step": 425
},
{
"epoch": 0.6012024048096193,
"grad_norm": 19.90398597717285,
"learning_rate": 1.6003560302625723e-05,
"loss": 3.0515,
"step": 450
},
{
"epoch": 0.6346025384101537,
"grad_norm": 25.174152374267578,
"learning_rate": 1.5781041388518027e-05,
"loss": 2.8242,
"step": 475
},
{
"epoch": 0.6680026720106881,
"grad_norm": 23.089248657226562,
"learning_rate": 1.5558522474410327e-05,
"loss": 2.7323,
"step": 500
},
{
"epoch": 0.7014028056112225,
"grad_norm": 31.02774429321289,
"learning_rate": 1.5336003560302627e-05,
"loss": 2.5633,
"step": 525
},
{
"epoch": 0.7348029392117569,
"grad_norm": 22.90081024169922,
"learning_rate": 1.5113484646194927e-05,
"loss": 2.5597,
"step": 550
},
{
"epoch": 0.7682030728122913,
"grad_norm": 18.37523078918457,
"learning_rate": 1.4890965732087228e-05,
"loss": 2.4584,
"step": 575
},
{
"epoch": 0.8016032064128257,
"grad_norm": 19.337141036987305,
"learning_rate": 1.466844681797953e-05,
"loss": 2.5439,
"step": 600
},
{
"epoch": 0.8350033400133601,
"grad_norm": 21.656770706176758,
"learning_rate": 1.444592790387183e-05,
"loss": 2.4629,
"step": 625
},
{
"epoch": 0.8684034736138945,
"grad_norm": 19.98714256286621,
"learning_rate": 1.4223408989764132e-05,
"loss": 2.4676,
"step": 650
},
{
"epoch": 0.9018036072144289,
"grad_norm": 29.199106216430664,
"learning_rate": 1.4000890075656433e-05,
"loss": 2.2897,
"step": 675
},
{
"epoch": 0.9352037408149633,
"grad_norm": 23.988719940185547,
"learning_rate": 1.3778371161548733e-05,
"loss": 2.1067,
"step": 700
},
{
"epoch": 0.9686038744154977,
"grad_norm": 18.715688705444336,
"learning_rate": 1.3555852247441033e-05,
"loss": 2.3692,
"step": 725
},
{
"epoch": 1.0013360053440215,
"grad_norm": 21.375459671020508,
"learning_rate": 1.3333333333333333e-05,
"loss": 2.1564,
"step": 750
},
{
"epoch": 1.0347361389445557,
"grad_norm": 9.197733879089355,
"learning_rate": 1.3110814419225635e-05,
"loss": 1.3881,
"step": 775
},
{
"epoch": 1.0681362725450902,
"grad_norm": 14.479204177856445,
"learning_rate": 1.2888295505117937e-05,
"loss": 1.4377,
"step": 800
}
],
"logging_steps": 25,
"max_steps": 2247,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.6585066994883625e+18,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}