{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.0406124355277586, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.000812248710555172, "grad_norm": 0.3937675654888153, "learning_rate": 0.00018, "loss": 1.2718635559082032, "step": 10 }, { "epoch": 0.001624497421110344, "grad_norm": 0.15561772882938385, "learning_rate": 0.00019939799331103678, "loss": 0.79568510055542, "step": 20 }, { "epoch": 0.0024367461316655158, "grad_norm": 0.1349518895149231, "learning_rate": 0.00019872909698996657, "loss": 0.6877449035644532, "step": 30 }, { "epoch": 0.003248994842220688, "grad_norm": 0.1342487335205078, "learning_rate": 0.00019806020066889633, "loss": 0.7130545616149903, "step": 40 }, { "epoch": 0.00406124355277586, "grad_norm": 0.11474266648292542, "learning_rate": 0.0001973913043478261, "loss": 0.6835604190826416, "step": 50 }, { "epoch": 0.0048734922633310316, "grad_norm": 0.10755404084920883, "learning_rate": 0.00019672240802675586, "loss": 0.652743148803711, "step": 60 }, { "epoch": 0.005685740973886204, "grad_norm": 0.11963178217411041, "learning_rate": 0.00019605351170568562, "loss": 0.6648129940032959, "step": 70 }, { "epoch": 0.006497989684441376, "grad_norm": 0.0965123176574707, "learning_rate": 0.0001953846153846154, "loss": 0.6699862003326416, "step": 80 }, { "epoch": 0.007310238394996548, "grad_norm": 0.10343202203512192, "learning_rate": 0.00019471571906354515, "loss": 0.7039784908294677, "step": 90 }, { "epoch": 0.00812248710555172, "grad_norm": 0.12766548991203308, "learning_rate": 0.00019404682274247492, "loss": 0.6552480697631836, "step": 100 }, { "epoch": 0.008934735816106891, "grad_norm": 0.10638143867254257, "learning_rate": 0.00019337792642140468, "loss": 0.6422908782958985, "step": 110 }, { "epoch": 0.009746984526662063, "grad_norm": 0.13318222761154175, "learning_rate": 0.00019270903010033444, "loss": 0.6597362995147705, "step": 120 }, { "epoch": 0.010559233237217237, "grad_norm": 0.11342291533946991, "learning_rate": 0.0001920401337792642, "loss": 0.6484825611114502, "step": 130 }, { "epoch": 0.011371481947772408, "grad_norm": 0.10095760971307755, "learning_rate": 0.000191371237458194, "loss": 0.6201153755187988, "step": 140 }, { "epoch": 0.01218373065832758, "grad_norm": 0.12489528954029083, "learning_rate": 0.00019070234113712376, "loss": 0.6547854900360107, "step": 150 }, { "epoch": 0.012995979368882752, "grad_norm": 0.10871083289384842, "learning_rate": 0.00019003344481605353, "loss": 0.654566764831543, "step": 160 }, { "epoch": 0.013808228079437924, "grad_norm": 0.1039392426609993, "learning_rate": 0.0001893645484949833, "loss": 0.649402379989624, "step": 170 }, { "epoch": 0.014620476789993096, "grad_norm": 0.1328577846288681, "learning_rate": 0.00018869565217391305, "loss": 0.6331918716430665, "step": 180 }, { "epoch": 0.015432725500548267, "grad_norm": 0.1274302750825882, "learning_rate": 0.00018802675585284282, "loss": 0.6307297706604004, "step": 190 }, { "epoch": 0.01624497421110344, "grad_norm": 0.12753233313560486, "learning_rate": 0.00018735785953177258, "loss": 0.6115467071533203, "step": 200 }, { "epoch": 0.01705722292165861, "grad_norm": 0.10285915434360504, "learning_rate": 0.00018668896321070235, "loss": 0.6152299880981446, "step": 210 }, { "epoch": 0.017869471632213783, "grad_norm": 0.17819221317768097, "learning_rate": 0.0001860200668896321, "loss": 0.6279479026794433, "step": 220 }, { "epoch": 0.018681720342768954, "grad_norm": 0.11906374990940094, "learning_rate": 0.00018535117056856187, "loss": 0.642233943939209, "step": 230 }, { "epoch": 0.019493969053324126, "grad_norm": 0.13238145411014557, "learning_rate": 0.00018468227424749164, "loss": 0.6336838245391846, "step": 240 }, { "epoch": 0.0203062177638793, "grad_norm": 0.11965897679328918, "learning_rate": 0.00018401337792642143, "loss": 0.6414345264434814, "step": 250 }, { "epoch": 0.021118466474434473, "grad_norm": 0.11781352013349533, "learning_rate": 0.0001833444816053512, "loss": 0.6286166191101075, "step": 260 }, { "epoch": 0.021930715184989645, "grad_norm": 0.13107237219810486, "learning_rate": 0.00018267558528428096, "loss": 0.6182798385620117, "step": 270 }, { "epoch": 0.022742963895544817, "grad_norm": 0.11255916953086853, "learning_rate": 0.00018200668896321072, "loss": 0.6299595832824707, "step": 280 }, { "epoch": 0.02355521260609999, "grad_norm": 0.11259932816028595, "learning_rate": 0.00018133779264214048, "loss": 0.6332510471343994, "step": 290 }, { "epoch": 0.02436746131665516, "grad_norm": 0.11110201478004456, "learning_rate": 0.00018066889632107025, "loss": 0.6277278900146485, "step": 300 }, { "epoch": 0.025179710027210332, "grad_norm": 0.11659885942935944, "learning_rate": 0.00018, "loss": 0.6012191772460938, "step": 310 }, { "epoch": 0.025991958737765504, "grad_norm": 0.11534283310174942, "learning_rate": 0.00017933110367892978, "loss": 0.6344992160797119, "step": 320 }, { "epoch": 0.026804207448320676, "grad_norm": 0.13291247189044952, "learning_rate": 0.00017866220735785954, "loss": 0.6238264560699462, "step": 330 }, { "epoch": 0.027616456158875848, "grad_norm": 0.12442389875650406, "learning_rate": 0.0001779933110367893, "loss": 0.6141870975494385, "step": 340 }, { "epoch": 0.02842870486943102, "grad_norm": 0.10461113601922989, "learning_rate": 0.00017732441471571907, "loss": 0.6277643203735351, "step": 350 }, { "epoch": 0.02924095357998619, "grad_norm": 0.11059945821762085, "learning_rate": 0.00017665551839464886, "loss": 0.650815773010254, "step": 360 }, { "epoch": 0.030053202290541363, "grad_norm": 0.1186944767832756, "learning_rate": 0.00017598662207357862, "loss": 0.6423202991485596, "step": 370 }, { "epoch": 0.030865451001096535, "grad_norm": 0.1306525617837906, "learning_rate": 0.00017531772575250839, "loss": 0.6001749038696289, "step": 380 }, { "epoch": 0.031677699711651706, "grad_norm": 0.12819500267505646, "learning_rate": 0.00017464882943143815, "loss": 0.6060415744781494, "step": 390 }, { "epoch": 0.03248994842220688, "grad_norm": 0.13138507306575775, "learning_rate": 0.0001739799331103679, "loss": 0.5925205707550049, "step": 400 }, { "epoch": 0.03330219713276205, "grad_norm": 0.14968585968017578, "learning_rate": 0.00017331103678929768, "loss": 0.6501868724822998, "step": 410 }, { "epoch": 0.03411444584331722, "grad_norm": 0.12787474691867828, "learning_rate": 0.00017264214046822744, "loss": 0.6108675956726074, "step": 420 }, { "epoch": 0.034926694553872394, "grad_norm": 0.13028663396835327, "learning_rate": 0.0001719732441471572, "loss": 0.6132050514221191, "step": 430 }, { "epoch": 0.035738943264427565, "grad_norm": 0.1314769834280014, "learning_rate": 0.00017130434782608697, "loss": 0.6264569759368896, "step": 440 }, { "epoch": 0.03655119197498274, "grad_norm": 0.11964884400367737, "learning_rate": 0.00017063545150501673, "loss": 0.6445899486541748, "step": 450 }, { "epoch": 0.03736344068553791, "grad_norm": 0.11594310402870178, "learning_rate": 0.0001699665551839465, "loss": 0.6022776603698731, "step": 460 }, { "epoch": 0.03817568939609308, "grad_norm": 0.11167627573013306, "learning_rate": 0.00016929765886287626, "loss": 0.6172833442687988, "step": 470 }, { "epoch": 0.03898793810664825, "grad_norm": 0.131842702627182, "learning_rate": 0.00016862876254180602, "loss": 0.6316081523895264, "step": 480 }, { "epoch": 0.039800186817203424, "grad_norm": 0.12949836254119873, "learning_rate": 0.0001679598662207358, "loss": 0.5979058742523193, "step": 490 }, { "epoch": 0.0406124355277586, "grad_norm": 0.1311161369085312, "learning_rate": 0.00016729096989966555, "loss": 0.6085984230041503, "step": 500 } ], "logging_steps": 10, "max_steps": 3000, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 3.351969772278866e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }