flutter-forge-model / checkpoint-1000 /trainer_state.json
abo003516's picture
Upload folder using huggingface_hub
f6cd61b verified
Raw
History Blame Contribute Delete
19.5 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.0812248710555172,
"eval_steps": 500,
"global_step": 1000,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.000812248710555172,
"grad_norm": 0.3937675654888153,
"learning_rate": 0.00018,
"loss": 1.2718635559082032,
"step": 10
},
{
"epoch": 0.001624497421110344,
"grad_norm": 0.15561772882938385,
"learning_rate": 0.00019939799331103678,
"loss": 0.79568510055542,
"step": 20
},
{
"epoch": 0.0024367461316655158,
"grad_norm": 0.1349518895149231,
"learning_rate": 0.00019872909698996657,
"loss": 0.6877449035644532,
"step": 30
},
{
"epoch": 0.003248994842220688,
"grad_norm": 0.1342487335205078,
"learning_rate": 0.00019806020066889633,
"loss": 0.7130545616149903,
"step": 40
},
{
"epoch": 0.00406124355277586,
"grad_norm": 0.11474266648292542,
"learning_rate": 0.0001973913043478261,
"loss": 0.6835604190826416,
"step": 50
},
{
"epoch": 0.0048734922633310316,
"grad_norm": 0.10755404084920883,
"learning_rate": 0.00019672240802675586,
"loss": 0.652743148803711,
"step": 60
},
{
"epoch": 0.005685740973886204,
"grad_norm": 0.11963178217411041,
"learning_rate": 0.00019605351170568562,
"loss": 0.6648129940032959,
"step": 70
},
{
"epoch": 0.006497989684441376,
"grad_norm": 0.0965123176574707,
"learning_rate": 0.0001953846153846154,
"loss": 0.6699862003326416,
"step": 80
},
{
"epoch": 0.007310238394996548,
"grad_norm": 0.10343202203512192,
"learning_rate": 0.00019471571906354515,
"loss": 0.7039784908294677,
"step": 90
},
{
"epoch": 0.00812248710555172,
"grad_norm": 0.12766548991203308,
"learning_rate": 0.00019404682274247492,
"loss": 0.6552480697631836,
"step": 100
},
{
"epoch": 0.008934735816106891,
"grad_norm": 0.10638143867254257,
"learning_rate": 0.00019337792642140468,
"loss": 0.6422908782958985,
"step": 110
},
{
"epoch": 0.009746984526662063,
"grad_norm": 0.13318222761154175,
"learning_rate": 0.00019270903010033444,
"loss": 0.6597362995147705,
"step": 120
},
{
"epoch": 0.010559233237217237,
"grad_norm": 0.11342291533946991,
"learning_rate": 0.0001920401337792642,
"loss": 0.6484825611114502,
"step": 130
},
{
"epoch": 0.011371481947772408,
"grad_norm": 0.10095760971307755,
"learning_rate": 0.000191371237458194,
"loss": 0.6201153755187988,
"step": 140
},
{
"epoch": 0.01218373065832758,
"grad_norm": 0.12489528954029083,
"learning_rate": 0.00019070234113712376,
"loss": 0.6547854900360107,
"step": 150
},
{
"epoch": 0.012995979368882752,
"grad_norm": 0.10871083289384842,
"learning_rate": 0.00019003344481605353,
"loss": 0.654566764831543,
"step": 160
},
{
"epoch": 0.013808228079437924,
"grad_norm": 0.1039392426609993,
"learning_rate": 0.0001893645484949833,
"loss": 0.649402379989624,
"step": 170
},
{
"epoch": 0.014620476789993096,
"grad_norm": 0.1328577846288681,
"learning_rate": 0.00018869565217391305,
"loss": 0.6331918716430665,
"step": 180
},
{
"epoch": 0.015432725500548267,
"grad_norm": 0.1274302750825882,
"learning_rate": 0.00018802675585284282,
"loss": 0.6307297706604004,
"step": 190
},
{
"epoch": 0.01624497421110344,
"grad_norm": 0.12753233313560486,
"learning_rate": 0.00018735785953177258,
"loss": 0.6115467071533203,
"step": 200
},
{
"epoch": 0.01705722292165861,
"grad_norm": 0.10285915434360504,
"learning_rate": 0.00018668896321070235,
"loss": 0.6152299880981446,
"step": 210
},
{
"epoch": 0.017869471632213783,
"grad_norm": 0.17819221317768097,
"learning_rate": 0.0001860200668896321,
"loss": 0.6279479026794433,
"step": 220
},
{
"epoch": 0.018681720342768954,
"grad_norm": 0.11906374990940094,
"learning_rate": 0.00018535117056856187,
"loss": 0.642233943939209,
"step": 230
},
{
"epoch": 0.019493969053324126,
"grad_norm": 0.13238145411014557,
"learning_rate": 0.00018468227424749164,
"loss": 0.6336838245391846,
"step": 240
},
{
"epoch": 0.0203062177638793,
"grad_norm": 0.11965897679328918,
"learning_rate": 0.00018401337792642143,
"loss": 0.6414345264434814,
"step": 250
},
{
"epoch": 0.021118466474434473,
"grad_norm": 0.11781352013349533,
"learning_rate": 0.0001833444816053512,
"loss": 0.6286166191101075,
"step": 260
},
{
"epoch": 0.021930715184989645,
"grad_norm": 0.13107237219810486,
"learning_rate": 0.00018267558528428096,
"loss": 0.6182798385620117,
"step": 270
},
{
"epoch": 0.022742963895544817,
"grad_norm": 0.11255916953086853,
"learning_rate": 0.00018200668896321072,
"loss": 0.6299595832824707,
"step": 280
},
{
"epoch": 0.02355521260609999,
"grad_norm": 0.11259932816028595,
"learning_rate": 0.00018133779264214048,
"loss": 0.6332510471343994,
"step": 290
},
{
"epoch": 0.02436746131665516,
"grad_norm": 0.11110201478004456,
"learning_rate": 0.00018066889632107025,
"loss": 0.6277278900146485,
"step": 300
},
{
"epoch": 0.025179710027210332,
"grad_norm": 0.11659885942935944,
"learning_rate": 0.00018,
"loss": 0.6012191772460938,
"step": 310
},
{
"epoch": 0.025991958737765504,
"grad_norm": 0.11534283310174942,
"learning_rate": 0.00017933110367892978,
"loss": 0.6344992160797119,
"step": 320
},
{
"epoch": 0.026804207448320676,
"grad_norm": 0.13291247189044952,
"learning_rate": 0.00017866220735785954,
"loss": 0.6238264560699462,
"step": 330
},
{
"epoch": 0.027616456158875848,
"grad_norm": 0.12442389875650406,
"learning_rate": 0.0001779933110367893,
"loss": 0.6141870975494385,
"step": 340
},
{
"epoch": 0.02842870486943102,
"grad_norm": 0.10461113601922989,
"learning_rate": 0.00017732441471571907,
"loss": 0.6277643203735351,
"step": 350
},
{
"epoch": 0.02924095357998619,
"grad_norm": 0.11059945821762085,
"learning_rate": 0.00017665551839464886,
"loss": 0.650815773010254,
"step": 360
},
{
"epoch": 0.030053202290541363,
"grad_norm": 0.1186944767832756,
"learning_rate": 0.00017598662207357862,
"loss": 0.6423202991485596,
"step": 370
},
{
"epoch": 0.030865451001096535,
"grad_norm": 0.1306525617837906,
"learning_rate": 0.00017531772575250839,
"loss": 0.6001749038696289,
"step": 380
},
{
"epoch": 0.031677699711651706,
"grad_norm": 0.12819500267505646,
"learning_rate": 0.00017464882943143815,
"loss": 0.6060415744781494,
"step": 390
},
{
"epoch": 0.03248994842220688,
"grad_norm": 0.13138507306575775,
"learning_rate": 0.0001739799331103679,
"loss": 0.5925205707550049,
"step": 400
},
{
"epoch": 0.03330219713276205,
"grad_norm": 0.14968585968017578,
"learning_rate": 0.00017331103678929768,
"loss": 0.6501868724822998,
"step": 410
},
{
"epoch": 0.03411444584331722,
"grad_norm": 0.12787474691867828,
"learning_rate": 0.00017264214046822744,
"loss": 0.6108675956726074,
"step": 420
},
{
"epoch": 0.034926694553872394,
"grad_norm": 0.13028663396835327,
"learning_rate": 0.0001719732441471572,
"loss": 0.6132050514221191,
"step": 430
},
{
"epoch": 0.035738943264427565,
"grad_norm": 0.1314769834280014,
"learning_rate": 0.00017130434782608697,
"loss": 0.6264569759368896,
"step": 440
},
{
"epoch": 0.03655119197498274,
"grad_norm": 0.11964884400367737,
"learning_rate": 0.00017063545150501673,
"loss": 0.6445899486541748,
"step": 450
},
{
"epoch": 0.03736344068553791,
"grad_norm": 0.11594310402870178,
"learning_rate": 0.0001699665551839465,
"loss": 0.6022776603698731,
"step": 460
},
{
"epoch": 0.03817568939609308,
"grad_norm": 0.11167627573013306,
"learning_rate": 0.00016929765886287626,
"loss": 0.6172833442687988,
"step": 470
},
{
"epoch": 0.03898793810664825,
"grad_norm": 0.131842702627182,
"learning_rate": 0.00016862876254180602,
"loss": 0.6316081523895264,
"step": 480
},
{
"epoch": 0.039800186817203424,
"grad_norm": 0.12949836254119873,
"learning_rate": 0.0001679598662207358,
"loss": 0.5979058742523193,
"step": 490
},
{
"epoch": 0.0406124355277586,
"grad_norm": 0.1311161369085312,
"learning_rate": 0.00016729096989966555,
"loss": 0.6085984230041503,
"step": 500
},
{
"epoch": 0.041424684238313775,
"grad_norm": 0.11565863341093063,
"learning_rate": 0.00016662207357859532,
"loss": 0.5864393234252929,
"step": 510
},
{
"epoch": 0.04223693294886895,
"grad_norm": 0.14647597074508667,
"learning_rate": 0.00016595317725752508,
"loss": 0.565390157699585,
"step": 520
},
{
"epoch": 0.04304918165942412,
"grad_norm": 0.11932296305894852,
"learning_rate": 0.00016528428093645484,
"loss": 0.6000445365905762,
"step": 530
},
{
"epoch": 0.04386143036997929,
"grad_norm": 0.14222687482833862,
"learning_rate": 0.0001646153846153846,
"loss": 0.6550194263458252,
"step": 540
},
{
"epoch": 0.04467367908053446,
"grad_norm": 0.12822918593883514,
"learning_rate": 0.00016394648829431437,
"loss": 0.6483684062957764,
"step": 550
},
{
"epoch": 0.045485927791089634,
"grad_norm": 0.13619735836982727,
"learning_rate": 0.00016327759197324413,
"loss": 0.6203540802001953,
"step": 560
},
{
"epoch": 0.046298176501644805,
"grad_norm": 0.13233502209186554,
"learning_rate": 0.00016260869565217393,
"loss": 0.6185849189758301,
"step": 570
},
{
"epoch": 0.04711042521219998,
"grad_norm": 0.12952451407909393,
"learning_rate": 0.0001619397993311037,
"loss": 0.5921258449554443,
"step": 580
},
{
"epoch": 0.04792267392275515,
"grad_norm": 0.12617622315883636,
"learning_rate": 0.00016127090301003345,
"loss": 0.6221595287322998,
"step": 590
},
{
"epoch": 0.04873492263331032,
"grad_norm": 0.10956388711929321,
"learning_rate": 0.00016060200668896322,
"loss": 0.599628210067749,
"step": 600
},
{
"epoch": 0.04954717134386549,
"grad_norm": 0.11363332718610764,
"learning_rate": 0.00015993311036789298,
"loss": 0.5865818977355957,
"step": 610
},
{
"epoch": 0.050359420054420664,
"grad_norm": 0.1252753585577011,
"learning_rate": 0.00015926421404682275,
"loss": 0.5941249847412109,
"step": 620
},
{
"epoch": 0.051171668764975836,
"grad_norm": 0.12859514355659485,
"learning_rate": 0.0001585953177257525,
"loss": 0.598096513748169,
"step": 630
},
{
"epoch": 0.05198391747553101,
"grad_norm": 0.11355341970920563,
"learning_rate": 0.00015792642140468227,
"loss": 0.6211743354797363,
"step": 640
},
{
"epoch": 0.05279616618608618,
"grad_norm": 0.12344721704721451,
"learning_rate": 0.00015725752508361204,
"loss": 0.5909175395965576,
"step": 650
},
{
"epoch": 0.05360841489664135,
"grad_norm": 0.1423984318971634,
"learning_rate": 0.0001565886287625418,
"loss": 0.5645668029785156,
"step": 660
},
{
"epoch": 0.05442066360719652,
"grad_norm": 0.11847086995840073,
"learning_rate": 0.00015591973244147156,
"loss": 0.6111773014068603,
"step": 670
},
{
"epoch": 0.055232912317751695,
"grad_norm": 0.10838636755943298,
"learning_rate": 0.00015525083612040136,
"loss": 0.598857069015503,
"step": 680
},
{
"epoch": 0.05604516102830687,
"grad_norm": 0.12422847002744675,
"learning_rate": 0.00015458193979933112,
"loss": 0.6086232185363769,
"step": 690
},
{
"epoch": 0.05685740973886204,
"grad_norm": 0.11844707280397415,
"learning_rate": 0.00015391304347826088,
"loss": 0.6083204269409179,
"step": 700
},
{
"epoch": 0.05766965844941721,
"grad_norm": 0.12434474378824234,
"learning_rate": 0.00015324414715719065,
"loss": 0.5938045024871826,
"step": 710
},
{
"epoch": 0.05848190715997238,
"grad_norm": 0.1373196393251419,
"learning_rate": 0.0001525752508361204,
"loss": 0.5924835205078125,
"step": 720
},
{
"epoch": 0.059294155870527554,
"grad_norm": 0.1279742270708084,
"learning_rate": 0.00015190635451505017,
"loss": 0.5970086574554443,
"step": 730
},
{
"epoch": 0.060106404581082726,
"grad_norm": 0.1276446133852005,
"learning_rate": 0.00015123745819397994,
"loss": 0.6073202610015869,
"step": 740
},
{
"epoch": 0.0609186532916379,
"grad_norm": 0.15001341700553894,
"learning_rate": 0.0001505685618729097,
"loss": 0.5822910785675048,
"step": 750
},
{
"epoch": 0.06173090200219307,
"grad_norm": 0.13396278023719788,
"learning_rate": 0.00014989966555183947,
"loss": 0.5936754226684571,
"step": 760
},
{
"epoch": 0.06254315071274824,
"grad_norm": 0.21268559992313385,
"learning_rate": 0.00014923076923076923,
"loss": 0.6138639926910401,
"step": 770
},
{
"epoch": 0.06335539942330341,
"grad_norm": 0.1419883519411087,
"learning_rate": 0.000148561872909699,
"loss": 0.6192237854003906,
"step": 780
},
{
"epoch": 0.06416764813385858,
"grad_norm": 0.12715154886245728,
"learning_rate": 0.00014789297658862879,
"loss": 0.6032827377319336,
"step": 790
},
{
"epoch": 0.06497989684441376,
"grad_norm": 0.14222534000873566,
"learning_rate": 0.00014722408026755855,
"loss": 0.5837327480316162,
"step": 800
},
{
"epoch": 0.06579214555496893,
"grad_norm": 0.12761105597019196,
"learning_rate": 0.0001465551839464883,
"loss": 0.5880168437957763,
"step": 810
},
{
"epoch": 0.0666043942655241,
"grad_norm": 0.13514447212219238,
"learning_rate": 0.00014588628762541808,
"loss": 0.6178320407867431,
"step": 820
},
{
"epoch": 0.06741664297607927,
"grad_norm": 0.1156405657529831,
"learning_rate": 0.00014521739130434784,
"loss": 0.5975133895874023,
"step": 830
},
{
"epoch": 0.06822889168663444,
"grad_norm": 0.13029585778713226,
"learning_rate": 0.0001445484949832776,
"loss": 0.6032638549804688,
"step": 840
},
{
"epoch": 0.06904114039718962,
"grad_norm": 0.12680095434188843,
"learning_rate": 0.00014387959866220737,
"loss": 0.5651222705841065,
"step": 850
},
{
"epoch": 0.06985338910774479,
"grad_norm": 0.14011035859584808,
"learning_rate": 0.00014321070234113713,
"loss": 0.5644700050354003,
"step": 860
},
{
"epoch": 0.07066563781829996,
"grad_norm": 0.12153138220310211,
"learning_rate": 0.0001425418060200669,
"loss": 0.5840569019317627,
"step": 870
},
{
"epoch": 0.07147788652885513,
"grad_norm": 0.13272343575954437,
"learning_rate": 0.00014187290969899666,
"loss": 0.6429564476013183,
"step": 880
},
{
"epoch": 0.0722901352394103,
"grad_norm": 0.14595189690589905,
"learning_rate": 0.00014120401337792642,
"loss": 0.5844541072845459,
"step": 890
},
{
"epoch": 0.07310238394996547,
"grad_norm": 0.1272500455379486,
"learning_rate": 0.00014053511705685621,
"loss": 0.6260814189910888,
"step": 900
},
{
"epoch": 0.07391463266052065,
"grad_norm": 0.1340891271829605,
"learning_rate": 0.00013986622073578598,
"loss": 0.5867656707763672,
"step": 910
},
{
"epoch": 0.07472688137107582,
"grad_norm": 0.14307370781898499,
"learning_rate": 0.00013919732441471574,
"loss": 0.577037000656128,
"step": 920
},
{
"epoch": 0.07553913008163099,
"grad_norm": 0.15650159120559692,
"learning_rate": 0.0001385284280936455,
"loss": 0.6175992488861084,
"step": 930
},
{
"epoch": 0.07635137879218616,
"grad_norm": 0.15327556431293488,
"learning_rate": 0.00013785953177257527,
"loss": 0.5842644691467285,
"step": 940
},
{
"epoch": 0.07716362750274133,
"grad_norm": 0.12911345064640045,
"learning_rate": 0.00013719063545150503,
"loss": 0.5888010025024414,
"step": 950
},
{
"epoch": 0.0779758762132965,
"grad_norm": 0.16856183111667633,
"learning_rate": 0.00013652173913043477,
"loss": 0.6148488521575928,
"step": 960
},
{
"epoch": 0.07878812492385168,
"grad_norm": 0.1216643676161766,
"learning_rate": 0.00013585284280936453,
"loss": 0.5580620765686035,
"step": 970
},
{
"epoch": 0.07960037363440685,
"grad_norm": 0.12570665776729584,
"learning_rate": 0.0001351839464882943,
"loss": 0.5656413078308106,
"step": 980
},
{
"epoch": 0.08041262234496203,
"grad_norm": 0.1392045021057129,
"learning_rate": 0.0001345150501672241,
"loss": 0.6019979000091553,
"step": 990
},
{
"epoch": 0.0812248710555172,
"grad_norm": 0.13696540892124176,
"learning_rate": 0.00013384615384615385,
"loss": 0.5685174465179443,
"step": 1000
}
],
"logging_steps": 10,
"max_steps": 3000,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 6.684342531125576e+17,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}