ember-sft / checkpoints /checkpoint-500 /trainer_state.json
Kush26's picture
Upload folder using huggingface_hub
58a8d37 verified
Raw
History Blame Contribute Delete
9.93 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.1417142857142857,
"eval_steps": 500,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.022857142857142857,
"grad_norm": 0.24084052443504333,
"learning_rate": 3.6e-05,
"loss": 1.8805160522460938,
"step": 10
},
{
"epoch": 0.045714285714285714,
"grad_norm": 0.7475871443748474,
"learning_rate": 7.6e-05,
"loss": 1.8558967590332032,
"step": 20
},
{
"epoch": 0.06857142857142857,
"grad_norm": 0.36147743463516235,
"learning_rate": 0.000116,
"loss": 1.7081331253051757,
"step": 30
},
{
"epoch": 0.09142857142857143,
"grad_norm": 0.20821920037269592,
"learning_rate": 0.00015600000000000002,
"loss": 1.7803277969360352,
"step": 40
},
{
"epoch": 0.11428571428571428,
"grad_norm": 0.263358473777771,
"learning_rate": 0.000196,
"loss": 1.766655158996582,
"step": 50
},
{
"epoch": 0.13714285714285715,
"grad_norm": 0.3143269419670105,
"learning_rate": 0.00019998948817948157,
"loss": 1.6175874710083007,
"step": 60
},
{
"epoch": 0.16,
"grad_norm": 0.33422476053237915,
"learning_rate": 0.0001999531538593893,
"loss": 1.7259719848632813,
"step": 70
},
{
"epoch": 0.18285714285714286,
"grad_norm": 0.22533851861953735,
"learning_rate": 0.00019989087669249037,
"loss": 1.4437444686889649,
"step": 80
},
{
"epoch": 0.2057142857142857,
"grad_norm": 0.3600349426269531,
"learning_rate": 0.00019980267284282717,
"loss": 1.5337361335754394,
"step": 90
},
{
"epoch": 0.22857142857142856,
"grad_norm": 0.2413090616464615,
"learning_rate": 0.0001996885652037138,
"loss": 1.5476616859436034,
"step": 100
},
{
"epoch": 0.25142857142857145,
"grad_norm": 0.2523901164531708,
"learning_rate": 0.00019954858339179464,
"loss": 1.5455998420715331,
"step": 110
},
{
"epoch": 0.2742857142857143,
"grad_norm": 0.24484537541866302,
"learning_rate": 0.00019938276373935687,
"loss": 1.513450527191162,
"step": 120
},
{
"epoch": 0.29714285714285715,
"grad_norm": 0.24838711321353912,
"learning_rate": 0.00019919114928490086,
"loss": 1.4492396354675292,
"step": 130
},
{
"epoch": 0.32,
"grad_norm": 0.2666800618171692,
"learning_rate": 0.0001989737897619692,
"loss": 1.482685661315918,
"step": 140
},
{
"epoch": 0.34285714285714286,
"grad_norm": 0.27357691526412964,
"learning_rate": 0.0001987307415862385,
"loss": 1.6114809036254882,
"step": 150
},
{
"epoch": 0.3657142857142857,
"grad_norm": 0.2718013525009155,
"learning_rate": 0.0001984620678408767,
"loss": 1.5441733360290528,
"step": 160
},
{
"epoch": 0.38857142857142857,
"grad_norm": 0.4157712161540985,
"learning_rate": 0.0001981678382601698,
"loss": 1.4818391799926758,
"step": 170
},
{
"epoch": 0.4114285714285714,
"grad_norm": 0.22732244431972504,
"learning_rate": 0.0001978481292114223,
"loss": 1.4110660552978516,
"step": 180
},
{
"epoch": 0.4342857142857143,
"grad_norm": 0.34649384021759033,
"learning_rate": 0.00019750302367513612,
"loss": 1.4578609466552734,
"step": 190
},
{
"epoch": 0.45714285714285713,
"grad_norm": 0.25781163573265076,
"learning_rate": 0.00019713261122347294,
"loss": 1.571424674987793,
"step": 200
},
{
"epoch": 0.48,
"grad_norm": 0.28305044770240784,
"learning_rate": 0.0001967369879970058,
"loss": 1.4585536003112793,
"step": 210
},
{
"epoch": 0.5028571428571429,
"grad_norm": 0.2838617265224457,
"learning_rate": 0.00019631625667976583,
"loss": 1.4340142250061034,
"step": 220
},
{
"epoch": 0.5257142857142857,
"grad_norm": 0.22012515366077423,
"learning_rate": 0.00019587052647259044,
"loss": 1.5476778030395508,
"step": 230
},
{
"epoch": 0.5485714285714286,
"grad_norm": 0.3474932014942169,
"learning_rate": 0.00019539991306478046,
"loss": 1.576882266998291,
"step": 240
},
{
"epoch": 0.5714285714285714,
"grad_norm": 0.33951514959335327,
"learning_rate": 0.00019490453860407278,
"loss": 1.328181838989258,
"step": 250
},
{
"epoch": 0.5942857142857143,
"grad_norm": 0.3432082235813141,
"learning_rate": 0.00019438453166493712,
"loss": 1.391934299468994,
"step": 260
},
{
"epoch": 0.6171428571428571,
"grad_norm": 0.28918349742889404,
"learning_rate": 0.0001938400272152042,
"loss": 1.502618408203125,
"step": 270
},
{
"epoch": 0.64,
"grad_norm": 0.25333958864212036,
"learning_rate": 0.00019327116658103525,
"loss": 1.38908109664917,
"step": 280
},
{
"epoch": 0.6628571428571428,
"grad_norm": 0.3463074266910553,
"learning_rate": 0.0001926780974102403,
"loss": 1.481637668609619,
"step": 290
},
{
"epoch": 0.6857142857142857,
"grad_norm": 0.35754460096359253,
"learning_rate": 0.0001920609736339567,
"loss": 1.4449010848999024,
"step": 300
},
{
"epoch": 0.7085714285714285,
"grad_norm": 0.2789507210254669,
"learning_rate": 0.0001914199554266958,
"loss": 1.3220888137817384,
"step": 310
},
{
"epoch": 0.7314285714285714,
"grad_norm": 0.29721778631210327,
"learning_rate": 0.0001907552091647701,
"loss": 1.3448283195495605,
"step": 320
},
{
"epoch": 0.7542857142857143,
"grad_norm": 0.2662886381149292,
"learning_rate": 0.00019006690738310989,
"loss": 1.4433917999267578,
"step": 330
},
{
"epoch": 0.7771428571428571,
"grad_norm": 0.3506792485713959,
"learning_rate": 0.000189355228730482,
"loss": 1.3629341125488281,
"step": 340
},
{
"epoch": 0.8,
"grad_norm": 0.3221069872379303,
"learning_rate": 0.00018862035792312147,
"loss": 1.4070910453796386,
"step": 350
},
{
"epoch": 0.8228571428571428,
"grad_norm": 0.20347386598587036,
"learning_rate": 0.00018786248569678846,
"loss": 1.27726411819458,
"step": 360
},
{
"epoch": 0.8457142857142858,
"grad_norm": 0.29632076621055603,
"learning_rate": 0.00018708180875726265,
"loss": 1.4044340133666993,
"step": 370
},
{
"epoch": 0.8685714285714285,
"grad_norm": 0.29475492238998413,
"learning_rate": 0.00018627852972928838,
"loss": 1.3279429435729981,
"step": 380
},
{
"epoch": 0.8914285714285715,
"grad_norm": 0.30218783020973206,
"learning_rate": 0.00018545285710398342,
"loss": 1.271165943145752,
"step": 390
},
{
"epoch": 0.9142857142857143,
"grad_norm": 0.39957311749458313,
"learning_rate": 0.00018460500518472487,
"loss": 1.3363965034484864,
"step": 400
},
{
"epoch": 0.9371428571428572,
"grad_norm": 0.35526371002197266,
"learning_rate": 0.00018373519403152696,
"loss": 1.3019601821899414,
"step": 410
},
{
"epoch": 0.96,
"grad_norm": 0.3040597438812256,
"learning_rate": 0.00018284364940392424,
"loss": 1.35850830078125,
"step": 420
},
{
"epoch": 0.9828571428571429,
"grad_norm": 0.3702254593372345,
"learning_rate": 0.00018193060270237595,
"loss": 1.4040640830993651,
"step": 430
},
{
"epoch": 1.0045714285714287,
"grad_norm": 0.26018935441970825,
"learning_rate": 0.00018099629090820562,
"loss": 1.3547307014465333,
"step": 440
},
{
"epoch": 1.0274285714285714,
"grad_norm": 0.40254613757133484,
"learning_rate": 0.00018004095652209302,
"loss": 1.287850570678711,
"step": 450
},
{
"epoch": 1.0502857142857143,
"grad_norm": 0.3126438558101654,
"learning_rate": 0.0001790648475011327,
"loss": 1.2667573928833007,
"step": 460
},
{
"epoch": 1.0731428571428572,
"grad_norm": 0.2844507694244385,
"learning_rate": 0.00017806821719447695,
"loss": 1.2534740447998047,
"step": 470
},
{
"epoch": 1.096,
"grad_norm": 0.3075398802757263,
"learning_rate": 0.00017705132427757895,
"loss": 1.22727689743042,
"step": 480
},
{
"epoch": 1.1188571428571428,
"grad_norm": 0.3089239299297333,
"learning_rate": 0.00017601443268505342,
"loss": 1.2206268310546875,
"step": 490
},
{
"epoch": 1.1417142857142857,
"grad_norm": 0.37757185101509094,
"learning_rate": 0.00017495781154217265,
"loss": 1.3226361274719238,
"step": 500
}
],
"logging_steps": 10,
"max_steps": 2000,
"num_input_tokens_seen": 0,
"num_train_epochs": 5,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.214412985698099e+16,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}