{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 449, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.022308979364194088, "grad_norm": 0.47332054376602173, "learning_rate": 0.0001998790061807298, "loss": 0.4195059299468994, "step": 10 }, { "epoch": 0.044617958728388175, "grad_norm": 0.6206803917884827, "learning_rate": 0.00019928708812698545, "loss": 0.36927917003631594, "step": 20 }, { "epoch": 0.06692693809258227, "grad_norm": 0.5743476748466492, "learning_rate": 0.00019820494141215104, "loss": 0.3785541296005249, "step": 30 }, { "epoch": 0.08923591745677635, "grad_norm": 0.49116021394729614, "learning_rate": 0.00019663790912106393, "loss": 0.36582486629486083, "step": 40 }, { "epoch": 0.11154489682097044, "grad_norm": 0.6498185992240906, "learning_rate": 0.00019459372845456705, "loss": 0.3734700679779053, "step": 50 }, { "epoch": 0.13385387618516453, "grad_norm": 0.6294810175895691, "learning_rate": 0.0001920824925271838, "loss": 0.3142746686935425, "step": 60 }, { "epoch": 0.1561628555493586, "grad_norm": 0.610817015171051, "learning_rate": 0.00018911660053250103, "loss": 0.3576608657836914, "step": 70 }, { "epoch": 0.1784718349135527, "grad_norm": 0.45995697379112244, "learning_rate": 0.0001857106965223177, "loss": 0.30390095710754395, "step": 80 }, { "epoch": 0.2007808142777468, "grad_norm": 1.2754241228103638, "learning_rate": 0.00018188159710183594, "loss": 0.3097202301025391, "step": 90 }, { "epoch": 0.22308979364194087, "grad_norm": 0.49120989441871643, "learning_rate": 0.00017764820839789964, "loss": 0.30510644912719725, "step": 100 }, { "epoch": 0.24539877300613497, "grad_norm": 0.45583832263946533, "learning_rate": 0.00017303143271024744, "loss": 0.2922208786010742, "step": 110 }, { "epoch": 0.26770775237032907, "grad_norm": 0.5388043522834778, "learning_rate": 0.0001680540653066891, "loss": 0.2974637508392334, "step": 120 }, { "epoch": 0.29001673173452314, "grad_norm": 0.49362650513648987, "learning_rate": 0.00016274068187177771, "loss": 0.2713587760925293, "step": 130 }, { "epoch": 0.3123257110987172, "grad_norm": 1.0019351243972778, "learning_rate": 0.00015711751716469786, "loss": 0.392174506187439, "step": 140 }, { "epoch": 0.33463469046291133, "grad_norm": 0.497901052236557, "learning_rate": 0.0001512123354854955, "loss": 0.29191443920135496, "step": 150 }, { "epoch": 0.3569436698271054, "grad_norm": 0.6082267761230469, "learning_rate": 0.00014505429358922, "loss": 0.3747562408447266, "step": 160 }, { "epoch": 0.3792526491912995, "grad_norm": 0.47960519790649414, "learning_rate": 0.0001386737967248388, "loss": 0.3827587842941284, "step": 170 }, { "epoch": 0.4015616285554936, "grad_norm": 0.619045078754425, "learning_rate": 0.00013210234850972964, "loss": 0.30683255195617676, "step": 180 }, { "epoch": 0.42387060791968767, "grad_norm": 0.5024956464767456, "learning_rate": 0.00012537239538099425, "loss": 0.34638049602508547, "step": 190 }, { "epoch": 0.44617958728388174, "grad_norm": 0.5191444158554077, "learning_rate": 0.00011851716639161159, "loss": 0.36590137481689455, "step": 200 }, { "epoch": 0.46848856664807587, "grad_norm": 0.36740896105766296, "learning_rate": 0.00011157050914243614, "loss": 0.31072416305541994, "step": 210 }, { "epoch": 0.49079754601226994, "grad_norm": 0.43472257256507874, "learning_rate": 0.00010456672266012446, "loss": 0.3114518880844116, "step": 220 }, { "epoch": 0.5131065253764641, "grad_norm": 0.4393475353717804, "learning_rate": 9.754038804615257e-05, "loss": 0.2944304943084717, "step": 230 }, { "epoch": 0.5354155047406581, "grad_norm": 0.4425562024116516, "learning_rate": 9.052619773309317e-05, "loss": 0.2811075925827026, "step": 240 }, { "epoch": 0.5577244841048522, "grad_norm": 0.4982526898384094, "learning_rate": 8.355878419119657e-05, "loss": 0.2992835521697998, "step": 250 }, { "epoch": 0.5800334634690463, "grad_norm": 1.5215120315551758, "learning_rate": 7.667254893103519e-05, "loss": 0.3270727634429932, "step": 260 }, { "epoch": 0.6023424428332403, "grad_norm": 0.4536781311035156, "learning_rate": 6.990149264650814e-05, "loss": 0.29621281623840334, "step": 270 }, { "epoch": 0.6246514221974344, "grad_norm": 0.47364547848701477, "learning_rate": 6.32790473368728e-05, "loss": 0.2747994661331177, "step": 280 }, { "epoch": 0.6469604015616286, "grad_norm": 0.4716944992542267, "learning_rate": 5.6837911236698536e-05, "loss": 0.3186234474182129, "step": 290 }, { "epoch": 0.6692693809258227, "grad_norm": 0.42809948325157166, "learning_rate": 5.060988736877366e-05, "loss": 0.28460078239440917, "step": 300 }, { "epoch": 0.6915783602900167, "grad_norm": 0.4650896191596985, "learning_rate": 4.462572651710847e-05, "loss": 0.30340597629547117, "step": 310 }, { "epoch": 0.7138873396542108, "grad_norm": 0.7915768623352051, "learning_rate": 3.8914975395353334e-05, "loss": 0.32752094268798826, "step": 320 }, { "epoch": 0.7361963190184049, "grad_norm": 0.5179622769355774, "learning_rate": 3.350583076029754e-05, "loss": 0.3099134206771851, "step": 330 }, { "epoch": 0.758505298382599, "grad_norm": 0.6544927954673767, "learning_rate": 2.8425000190762353e-05, "loss": 0.3012081623077393, "step": 340 }, { "epoch": 0.7808142777467931, "grad_norm": 0.4660824239253998, "learning_rate": 2.3697570219290077e-05, "loss": 0.26200897693634034, "step": 350 }, { "epoch": 0.8031232571109872, "grad_norm": 0.6966566443443298, "learning_rate": 1.9346882467727325e-05, "loss": 0.3476716041564941, "step": 360 }, { "epoch": 0.8254322364751813, "grad_norm": 0.5577861070632935, "learning_rate": 1.5394418398281352e-05, "loss": 0.3011465549468994, "step": 370 }, { "epoch": 0.8477412158393753, "grad_norm": 0.559313952922821, "learning_rate": 1.1859693249089642e-05, "loss": 0.2599821090698242, "step": 380 }, { "epoch": 0.8700501952035694, "grad_norm": 0.3570460379123688, "learning_rate": 8.760159677994172e-06, "loss": 0.2864186763763428, "step": 390 }, { "epoch": 0.8923591745677635, "grad_norm": 0.5391319990158081, "learning_rate": 6.111121590278346e-06, "loss": 0.30640947818756104, "step": 400 }, { "epoch": 0.9146681539319577, "grad_norm": 0.3447037637233734, "learning_rate": 3.925658575840696e-06, "loss": 0.29050464630126954, "step": 410 }, { "epoch": 0.9369771332961517, "grad_norm": 0.37274837493896484, "learning_rate": 2.2145613288957478e-06, "loss": 0.2897591829299927, "step": 420 }, { "epoch": 0.9592861126603458, "grad_norm": 0.43291178345680237, "learning_rate": 9.862783690666178e-07, "loss": 0.25869801044464114, "step": 430 }, { "epoch": 0.9815950920245399, "grad_norm": 0.7033930420875549, "learning_rate": 2.468743269331442e-07, "loss": 0.2812113046646118, "step": 440 } ], "logging_steps": 10, "max_steps": 449, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 7296295329092160.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }