| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.1417142857142857, |
| "eval_steps": 500, |
| "global_step": 500, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.022857142857142857, |
| "grad_norm": 0.24084052443504333, |
| "learning_rate": 3.6e-05, |
| "loss": 1.8805160522460938, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.045714285714285714, |
| "grad_norm": 0.7475871443748474, |
| "learning_rate": 7.6e-05, |
| "loss": 1.8558967590332032, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.06857142857142857, |
| "grad_norm": 0.36147743463516235, |
| "learning_rate": 0.000116, |
| "loss": 1.7081331253051757, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.09142857142857143, |
| "grad_norm": 0.20821920037269592, |
| "learning_rate": 0.00015600000000000002, |
| "loss": 1.7803277969360352, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.11428571428571428, |
| "grad_norm": 0.263358473777771, |
| "learning_rate": 0.000196, |
| "loss": 1.766655158996582, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.13714285714285715, |
| "grad_norm": 0.3143269419670105, |
| "learning_rate": 0.00019998948817948157, |
| "loss": 1.6175874710083007, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 0.33422476053237915, |
| "learning_rate": 0.0001999531538593893, |
| "loss": 1.7259719848632813, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.18285714285714286, |
| "grad_norm": 0.22533851861953735, |
| "learning_rate": 0.00019989087669249037, |
| "loss": 1.4437444686889649, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.2057142857142857, |
| "grad_norm": 0.3600349426269531, |
| "learning_rate": 0.00019980267284282717, |
| "loss": 1.5337361335754394, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.22857142857142856, |
| "grad_norm": 0.2413090616464615, |
| "learning_rate": 0.0001996885652037138, |
| "loss": 1.5476616859436034, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.25142857142857145, |
| "grad_norm": 0.2523901164531708, |
| "learning_rate": 0.00019954858339179464, |
| "loss": 1.5455998420715331, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.2742857142857143, |
| "grad_norm": 0.24484537541866302, |
| "learning_rate": 0.00019938276373935687, |
| "loss": 1.513450527191162, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.29714285714285715, |
| "grad_norm": 0.24838711321353912, |
| "learning_rate": 0.00019919114928490086, |
| "loss": 1.4492396354675292, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.32, |
| "grad_norm": 0.2666800618171692, |
| "learning_rate": 0.0001989737897619692, |
| "loss": 1.482685661315918, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.34285714285714286, |
| "grad_norm": 0.27357691526412964, |
| "learning_rate": 0.0001987307415862385, |
| "loss": 1.6114809036254882, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.3657142857142857, |
| "grad_norm": 0.2718013525009155, |
| "learning_rate": 0.0001984620678408767, |
| "loss": 1.5441733360290528, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.38857142857142857, |
| "grad_norm": 0.4157712161540985, |
| "learning_rate": 0.0001981678382601698, |
| "loss": 1.4818391799926758, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.4114285714285714, |
| "grad_norm": 0.22732244431972504, |
| "learning_rate": 0.0001978481292114223, |
| "loss": 1.4110660552978516, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.4342857142857143, |
| "grad_norm": 0.34649384021759033, |
| "learning_rate": 0.00019750302367513612, |
| "loss": 1.4578609466552734, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.45714285714285713, |
| "grad_norm": 0.25781163573265076, |
| "learning_rate": 0.00019713261122347294, |
| "loss": 1.571424674987793, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.48, |
| "grad_norm": 0.28305044770240784, |
| "learning_rate": 0.0001967369879970058, |
| "loss": 1.4585536003112793, |
| "step": 210 |
| }, |
| { |
| "epoch": 0.5028571428571429, |
| "grad_norm": 0.2838617265224457, |
| "learning_rate": 0.00019631625667976583, |
| "loss": 1.4340142250061034, |
| "step": 220 |
| }, |
| { |
| "epoch": 0.5257142857142857, |
| "grad_norm": 0.22012515366077423, |
| "learning_rate": 0.00019587052647259044, |
| "loss": 1.5476778030395508, |
| "step": 230 |
| }, |
| { |
| "epoch": 0.5485714285714286, |
| "grad_norm": 0.3474932014942169, |
| "learning_rate": 0.00019539991306478046, |
| "loss": 1.576882266998291, |
| "step": 240 |
| }, |
| { |
| "epoch": 0.5714285714285714, |
| "grad_norm": 0.33951514959335327, |
| "learning_rate": 0.00019490453860407278, |
| "loss": 1.328181838989258, |
| "step": 250 |
| }, |
| { |
| "epoch": 0.5942857142857143, |
| "grad_norm": 0.3432082235813141, |
| "learning_rate": 0.00019438453166493712, |
| "loss": 1.391934299468994, |
| "step": 260 |
| }, |
| { |
| "epoch": 0.6171428571428571, |
| "grad_norm": 0.28918349742889404, |
| "learning_rate": 0.0001938400272152042, |
| "loss": 1.502618408203125, |
| "step": 270 |
| }, |
| { |
| "epoch": 0.64, |
| "grad_norm": 0.25333958864212036, |
| "learning_rate": 0.00019327116658103525, |
| "loss": 1.38908109664917, |
| "step": 280 |
| }, |
| { |
| "epoch": 0.6628571428571428, |
| "grad_norm": 0.3463074266910553, |
| "learning_rate": 0.0001926780974102403, |
| "loss": 1.481637668609619, |
| "step": 290 |
| }, |
| { |
| "epoch": 0.6857142857142857, |
| "grad_norm": 0.35754460096359253, |
| "learning_rate": 0.0001920609736339567, |
| "loss": 1.4449010848999024, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.7085714285714285, |
| "grad_norm": 0.2789507210254669, |
| "learning_rate": 0.0001914199554266958, |
| "loss": 1.3220888137817384, |
| "step": 310 |
| }, |
| { |
| "epoch": 0.7314285714285714, |
| "grad_norm": 0.29721778631210327, |
| "learning_rate": 0.0001907552091647701, |
| "loss": 1.3448283195495605, |
| "step": 320 |
| }, |
| { |
| "epoch": 0.7542857142857143, |
| "grad_norm": 0.2662886381149292, |
| "learning_rate": 0.00019006690738310989, |
| "loss": 1.4433917999267578, |
| "step": 330 |
| }, |
| { |
| "epoch": 0.7771428571428571, |
| "grad_norm": 0.3506792485713959, |
| "learning_rate": 0.000189355228730482, |
| "loss": 1.3629341125488281, |
| "step": 340 |
| }, |
| { |
| "epoch": 0.8, |
| "grad_norm": 0.3221069872379303, |
| "learning_rate": 0.00018862035792312147, |
| "loss": 1.4070910453796386, |
| "step": 350 |
| }, |
| { |
| "epoch": 0.8228571428571428, |
| "grad_norm": 0.20347386598587036, |
| "learning_rate": 0.00018786248569678846, |
| "loss": 1.27726411819458, |
| "step": 360 |
| }, |
| { |
| "epoch": 0.8457142857142858, |
| "grad_norm": 0.29632076621055603, |
| "learning_rate": 0.00018708180875726265, |
| "loss": 1.4044340133666993, |
| "step": 370 |
| }, |
| { |
| "epoch": 0.8685714285714285, |
| "grad_norm": 0.29475492238998413, |
| "learning_rate": 0.00018627852972928838, |
| "loss": 1.3279429435729981, |
| "step": 380 |
| }, |
| { |
| "epoch": 0.8914285714285715, |
| "grad_norm": 0.30218783020973206, |
| "learning_rate": 0.00018545285710398342, |
| "loss": 1.271165943145752, |
| "step": 390 |
| }, |
| { |
| "epoch": 0.9142857142857143, |
| "grad_norm": 0.39957311749458313, |
| "learning_rate": 0.00018460500518472487, |
| "loss": 1.3363965034484864, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.9371428571428572, |
| "grad_norm": 0.35526371002197266, |
| "learning_rate": 0.00018373519403152696, |
| "loss": 1.3019601821899414, |
| "step": 410 |
| }, |
| { |
| "epoch": 0.96, |
| "grad_norm": 0.3040597438812256, |
| "learning_rate": 0.00018284364940392424, |
| "loss": 1.35850830078125, |
| "step": 420 |
| }, |
| { |
| "epoch": 0.9828571428571429, |
| "grad_norm": 0.3702254593372345, |
| "learning_rate": 0.00018193060270237595, |
| "loss": 1.4040640830993651, |
| "step": 430 |
| }, |
| { |
| "epoch": 1.0045714285714287, |
| "grad_norm": 0.26018935441970825, |
| "learning_rate": 0.00018099629090820562, |
| "loss": 1.3547307014465333, |
| "step": 440 |
| }, |
| { |
| "epoch": 1.0274285714285714, |
| "grad_norm": 0.40254613757133484, |
| "learning_rate": 0.00018004095652209302, |
| "loss": 1.287850570678711, |
| "step": 450 |
| }, |
| { |
| "epoch": 1.0502857142857143, |
| "grad_norm": 0.3126438558101654, |
| "learning_rate": 0.0001790648475011327, |
| "loss": 1.2667573928833007, |
| "step": 460 |
| }, |
| { |
| "epoch": 1.0731428571428572, |
| "grad_norm": 0.2844507694244385, |
| "learning_rate": 0.00017806821719447695, |
| "loss": 1.2534740447998047, |
| "step": 470 |
| }, |
| { |
| "epoch": 1.096, |
| "grad_norm": 0.3075398802757263, |
| "learning_rate": 0.00017705132427757895, |
| "loss": 1.22727689743042, |
| "step": 480 |
| }, |
| { |
| "epoch": 1.1188571428571428, |
| "grad_norm": 0.3089239299297333, |
| "learning_rate": 0.00017601443268505342, |
| "loss": 1.2206268310546875, |
| "step": 490 |
| }, |
| { |
| "epoch": 1.1417142857142857, |
| "grad_norm": 0.37757185101509094, |
| "learning_rate": 0.00017495781154217265, |
| "loss": 1.3226361274719238, |
| "step": 500 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 2000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 5, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 2.214412985698099e+16, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|