Quark-pre-30M / trainer_state.json
abdelkader-dev's picture
Upload 10 files
fee623f verified
Raw
History Blame Contribute Delete
21.8 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 10.0,
"eval_steps": 500,
"global_step": 11480,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.08710801393728224,
"grad_norm": 0.9188959002494812,
"learning_rate": 3.960000000000001e-05,
"loss": 6.191905517578125,
"step": 100
},
{
"epoch": 0.17421602787456447,
"grad_norm": 1.3060380220413208,
"learning_rate": 7.960000000000001e-05,
"loss": 6.121124877929687,
"step": 200
},
{
"epoch": 0.2613240418118467,
"grad_norm": 1.1203659772872925,
"learning_rate": 0.00011960000000000001,
"loss": 6.003787231445313,
"step": 300
},
{
"epoch": 0.34843205574912894,
"grad_norm": 1.421034812927246,
"learning_rate": 0.0001596,
"loss": 5.94820556640625,
"step": 400
},
{
"epoch": 0.4355400696864111,
"grad_norm": 1.2131978273391724,
"learning_rate": 0.0001996,
"loss": 5.933410034179688,
"step": 500
},
{
"epoch": 0.5226480836236934,
"grad_norm": 1.2755587100982666,
"learning_rate": 0.0001999598850347455,
"loss": 5.781154174804687,
"step": 600
},
{
"epoch": 0.6097560975609756,
"grad_norm": 1.4233345985412598,
"learning_rate": 0.0001998379481883983,
"loss": 5.668782958984375,
"step": 700
},
{
"epoch": 0.6968641114982579,
"grad_norm": 1.2652695178985596,
"learning_rate": 0.0001996342851837924,
"loss": 5.601710205078125,
"step": 800
},
{
"epoch": 0.7839721254355401,
"grad_norm": 1.1004737615585327,
"learning_rate": 0.0001993490627370437,
"loss": 5.515267944335937,
"step": 900
},
{
"epoch": 0.8710801393728222,
"grad_norm": 1.0849263668060303,
"learning_rate": 0.00019898251432785851,
"loss": 5.444522705078125,
"step": 1000
},
{
"epoch": 0.9581881533101045,
"grad_norm": 1.1359233856201172,
"learning_rate": 0.00019853494000840986,
"loss": 5.3566357421875,
"step": 1100
},
{
"epoch": 1.0452961672473868,
"grad_norm": 1.2210420370101929,
"learning_rate": 0.00019800670615771816,
"loss": 5.266632690429687,
"step": 1200
},
{
"epoch": 1.132404181184669,
"grad_norm": 1.4381468296051025,
"learning_rate": 0.00019739824518173792,
"loss": 5.174177856445312,
"step": 1300
},
{
"epoch": 1.2195121951219512,
"grad_norm": 1.0924814939498901,
"learning_rate": 0.00019671005515939518,
"loss": 5.1578131103515625,
"step": 1400
},
{
"epoch": 1.3066202090592334,
"grad_norm": 1.2317713499069214,
"learning_rate": 0.00019594269943486623,
"loss": 5.09080078125,
"step": 1500
},
{
"epoch": 1.3937282229965158,
"grad_norm": 1.1351029872894287,
"learning_rate": 0.00019509680615643058,
"loss": 5.07963623046875,
"step": 1600
},
{
"epoch": 1.480836236933798,
"grad_norm": 1.190399169921875,
"learning_rate": 0.00019417306776227622,
"loss": 5.024771118164063,
"step": 1700
},
{
"epoch": 1.5679442508710801,
"grad_norm": 1.2222275733947754,
"learning_rate": 0.00019317224041367817,
"loss": 5.00032470703125,
"step": 1800
},
{
"epoch": 1.6550522648083623,
"grad_norm": 1.535797119140625,
"learning_rate": 0.0001920951433760136,
"loss": 4.976044616699219,
"step": 1900
},
{
"epoch": 1.7421602787456445,
"grad_norm": 1.2563319206237793,
"learning_rate": 0.00019094265834812104,
"loss": 4.9138116455078125,
"step": 2000
},
{
"epoch": 1.8292682926829267,
"grad_norm": 1.526935338973999,
"learning_rate": 0.00018971572874055216,
"loss": 4.905592041015625,
"step": 2100
},
{
"epoch": 1.916376306620209,
"grad_norm": 1.355762004852295,
"learning_rate": 0.0001884153589033072,
"loss": 4.857882080078125,
"step": 2200
},
{
"epoch": 2.0034843205574915,
"grad_norm": 1.423733115196228,
"learning_rate": 0.00018704261330368583,
"loss": 4.808562316894531,
"step": 2300
},
{
"epoch": 2.0905923344947737,
"grad_norm": 1.6864680051803589,
"learning_rate": 0.00018559861565492692,
"loss": 4.654463806152344,
"step": 2400
},
{
"epoch": 2.177700348432056,
"grad_norm": 1.3302619457244873,
"learning_rate": 0.00018408454799635028,
"loss": 4.666855163574219,
"step": 2500
},
{
"epoch": 2.264808362369338,
"grad_norm": 1.3893911838531494,
"learning_rate": 0.00018250164972575327,
"loss": 4.609520568847656,
"step": 2600
},
{
"epoch": 2.35191637630662,
"grad_norm": 1.344896912574768,
"learning_rate": 0.0001808512165848545,
"loss": 4.636769409179688,
"step": 2700
},
{
"epoch": 2.4390243902439024,
"grad_norm": 1.274559497833252,
"learning_rate": 0.0001791345995986152,
"loss": 4.598162536621094,
"step": 2800
},
{
"epoch": 2.5261324041811846,
"grad_norm": 1.3674043416976929,
"learning_rate": 0.00017735320396930585,
"loss": 4.5604150390625,
"step": 2900
},
{
"epoch": 2.6132404181184667,
"grad_norm": 1.3470535278320312,
"learning_rate": 0.00017550848792622473,
"loss": 4.572581481933594,
"step": 3000
},
{
"epoch": 2.7003484320557494,
"grad_norm": 1.6086684465408325,
"learning_rate": 0.0001736019615320085,
"loss": 4.542828063964844,
"step": 3100
},
{
"epoch": 2.7874564459930316,
"grad_norm": 1.623887300491333,
"learning_rate": 0.00017163518544651283,
"loss": 4.519979858398438,
"step": 3200
},
{
"epoch": 2.8745644599303137,
"grad_norm": 1.4141863584518433,
"learning_rate": 0.00016960976964927507,
"loss": 4.497909240722656,
"step": 3300
},
{
"epoch": 2.961672473867596,
"grad_norm": 1.2833305597305298,
"learning_rate": 0.00016752737212160374,
"loss": 4.4761376953125,
"step": 3400
},
{
"epoch": 3.048780487804878,
"grad_norm": 1.5807830095291138,
"learning_rate": 0.000165389697489375,
"loss": 4.386007385253906,
"step": 3500
},
{
"epoch": 3.1358885017421603,
"grad_norm": 1.5675005912780762,
"learning_rate": 0.00016319849562764608,
"loss": 4.305965576171875,
"step": 3600
},
{
"epoch": 3.2229965156794425,
"grad_norm": 1.4463677406311035,
"learning_rate": 0.00016095556022822837,
"loss": 4.2965829467773435,
"step": 3700
},
{
"epoch": 3.3101045296167246,
"grad_norm": 1.495564341545105,
"learning_rate": 0.00015866272733139252,
"loss": 4.296761474609375,
"step": 3800
},
{
"epoch": 3.397212543554007,
"grad_norm": 1.6361560821533203,
"learning_rate": 0.0001563218738229078,
"loss": 4.255560607910156,
"step": 3900
},
{
"epoch": 3.484320557491289,
"grad_norm": 1.61375892162323,
"learning_rate": 0.00015393491589764567,
"loss": 4.2361297607421875,
"step": 4000
},
{
"epoch": 3.571428571428571,
"grad_norm": 1.710850715637207,
"learning_rate": 0.00015150380749100545,
"loss": 4.241318054199219,
"step": 4100
},
{
"epoch": 3.658536585365854,
"grad_norm": 1.888376235961914,
"learning_rate": 0.00014903053867944589,
"loss": 4.263135375976563,
"step": 4200
},
{
"epoch": 3.745644599303136,
"grad_norm": 1.5190894603729248,
"learning_rate": 0.00014651713405143234,
"loss": 4.203194580078125,
"step": 4300
},
{
"epoch": 3.832752613240418,
"grad_norm": 1.4314178228378296,
"learning_rate": 0.00014396565105013283,
"loss": 4.2452734375,
"step": 4400
},
{
"epoch": 3.9198606271777003,
"grad_norm": 1.4185148477554321,
"learning_rate": 0.0001413781782892191,
"loss": 4.218402099609375,
"step": 4500
},
{
"epoch": 4.006968641114983,
"grad_norm": 1.42833411693573,
"learning_rate": 0.00013875683384315278,
"loss": 4.195860290527344,
"step": 4600
},
{
"epoch": 4.094076655052265,
"grad_norm": 1.8272066116333008,
"learning_rate": 0.00013610376351335437,
"loss": 4.000545654296875,
"step": 4700
},
{
"epoch": 4.181184668989547,
"grad_norm": 1.6578880548477173,
"learning_rate": 0.00013342113907167596,
"loss": 4.044533386230468,
"step": 4800
},
{
"epoch": 4.2682926829268295,
"grad_norm": 1.719120740890503,
"learning_rate": 0.00013071115648261453,
"loss": 4.006611633300781,
"step": 4900
},
{
"epoch": 4.355400696864112,
"grad_norm": 1.8038350343704224,
"learning_rate": 0.00012797603410572144,
"loss": 4.005461120605469,
"step": 5000
},
{
"epoch": 4.442508710801394,
"grad_norm": 1.4871196746826172,
"learning_rate": 0.00012521801087967964,
"loss": 4.042769470214844,
"step": 5100
},
{
"epoch": 4.529616724738676,
"grad_norm": 1.4674580097198486,
"learning_rate": 0.00012243934448953522,
"loss": 4.014598999023438,
"step": 5200
},
{
"epoch": 4.616724738675958,
"grad_norm": 1.650697112083435,
"learning_rate": 0.00011964230951858309,
"loss": 4.0176239013671875,
"step": 5300
},
{
"epoch": 4.70383275261324,
"grad_norm": 2.049278497695923,
"learning_rate": 0.00011682919558642024,
"loss": 4.01257568359375,
"step": 5400
},
{
"epoch": 4.790940766550523,
"grad_norm": 1.789766550064087,
"learning_rate": 0.00011400230547469015,
"loss": 3.983531494140625,
"step": 5500
},
{
"epoch": 4.878048780487805,
"grad_norm": 1.9967801570892334,
"learning_rate": 0.0001111639532420534,
"loss": 3.9888790893554686,
"step": 5600
},
{
"epoch": 4.965156794425087,
"grad_norm": 1.5817842483520508,
"learning_rate": 0.00010831646232992643,
"loss": 3.9807525634765626,
"step": 5700
},
{
"epoch": 5.052264808362369,
"grad_norm": 1.8074733018875122,
"learning_rate": 0.00010546216366054025,
"loss": 3.868643798828125,
"step": 5800
},
{
"epoch": 5.139372822299651,
"grad_norm": 1.5427536964416504,
"learning_rate": 0.00010260339372887494,
"loss": 3.806876220703125,
"step": 5900
},
{
"epoch": 5.2264808362369335,
"grad_norm": 1.7108198404312134,
"learning_rate": 9.974249269003289e-05,
"loss": 3.824035949707031,
"step": 6000
},
{
"epoch": 5.313588850174216,
"grad_norm": 1.7547175884246826,
"learning_rate": 9.688180244361546e-05,
"loss": 3.8009918212890623,
"step": 6100
},
{
"epoch": 5.400696864111498,
"grad_norm": 2.0099358558654785,
"learning_rate": 9.40236647166718e-05,
"loss": 3.8154830932617188,
"step": 6200
},
{
"epoch": 5.487804878048781,
"grad_norm": 1.7973719835281372,
"learning_rate": 9.117041914678904e-05,
"loss": 3.8090631103515626,
"step": 6300
},
{
"epoch": 5.574912891986063,
"grad_norm": 1.7361223697662354,
"learning_rate": 8.832440136689258e-05,
"loss": 3.8208245849609375,
"step": 6400
},
{
"epoch": 5.662020905923345,
"grad_norm": 2.053532361984253,
"learning_rate": 8.548794109332481e-05,
"loss": 3.82370361328125,
"step": 6500
},
{
"epoch": 5.7491289198606275,
"grad_norm": 1.8259886503219604,
"learning_rate": 8.266336021876698e-05,
"loss": 3.826884460449219,
"step": 6600
},
{
"epoch": 5.83623693379791,
"grad_norm": 1.9447393417358398,
"learning_rate": 7.985297091156554e-05,
"loss": 3.8073544311523437,
"step": 6700
},
{
"epoch": 5.923344947735192,
"grad_norm": 1.7228244543075562,
"learning_rate": 7.70590737230184e-05,
"loss": 3.8064419555664064,
"step": 6800
},
{
"epoch": 6.010452961672474,
"grad_norm": 2.16105055809021,
"learning_rate": 7.428395570417108e-05,
"loss": 3.7939053344726563,
"step": 6900
},
{
"epoch": 6.097560975609756,
"grad_norm": 1.5277420282363892,
"learning_rate": 7.152988853366395e-05,
"loss": 3.6774615478515624,
"step": 7000
},
{
"epoch": 6.184668989547038,
"grad_norm": 1.8056442737579346,
"learning_rate": 6.879912665816299e-05,
"loss": 3.6514688110351563,
"step": 7100
},
{
"epoch": 6.2717770034843205,
"grad_norm": 1.9114874601364136,
"learning_rate": 6.60939054468966e-05,
"loss": 3.6658111572265626,
"step": 7200
},
{
"epoch": 6.358885017421603,
"grad_norm": 1.9075324535369873,
"learning_rate": 6.341643936180881e-05,
"loss": 3.6535992431640625,
"step": 7300
},
{
"epoch": 6.445993031358885,
"grad_norm": 1.952846646308899,
"learning_rate": 6.076892014482714e-05,
"loss": 3.6586459350585936,
"step": 7400
},
{
"epoch": 6.533101045296167,
"grad_norm": 1.8436208963394165,
"learning_rate": 5.8153515023728645e-05,
"loss": 3.6709304809570313,
"step": 7500
},
{
"epoch": 6.620209059233449,
"grad_norm": 1.5615363121032715,
"learning_rate": 5.5572364938073104e-05,
"loss": 3.6649066162109376,
"step": 7600
},
{
"epoch": 6.7073170731707314,
"grad_norm": 1.9237865209579468,
"learning_rate": 5.30275827866552e-05,
"loss": 3.6293704223632814,
"step": 7700
},
{
"epoch": 6.794425087108014,
"grad_norm": 1.8327174186706543,
"learning_rate": 5.052125169791064e-05,
"loss": 3.652876892089844,
"step": 7800
},
{
"epoch": 6.881533101045296,
"grad_norm": 1.786590576171875,
"learning_rate": 4.805542332469209e-05,
"loss": 3.655260009765625,
"step": 7900
},
{
"epoch": 6.968641114982578,
"grad_norm": 2.09110164642334,
"learning_rate": 4.563211616481061e-05,
"loss": 3.6624679565429688,
"step": 8000
},
{
"epoch": 7.055749128919861,
"grad_norm": 1.8738207817077637,
"learning_rate": 4.325331390871703e-05,
"loss": 3.606045837402344,
"step": 8100
},
{
"epoch": 7.142857142857143,
"grad_norm": 1.9863561391830444,
"learning_rate": 4.092096381567675e-05,
"loss": 3.5570523071289064,
"step": 8200
},
{
"epoch": 7.229965156794425,
"grad_norm": 2.195230722427368,
"learning_rate": 3.863697511976634e-05,
"loss": 3.5478726196289063,
"step": 8300
},
{
"epoch": 7.317073170731708,
"grad_norm": 1.8687387704849243,
"learning_rate": 3.640321746699732e-05,
"loss": 3.5580734252929687,
"step": 8400
},
{
"epoch": 7.40418118466899,
"grad_norm": 1.9148145914077759,
"learning_rate": 3.4221519384846014e-05,
"loss": 3.5321905517578127,
"step": 8500
},
{
"epoch": 7.491289198606272,
"grad_norm": 1.9626288414001465,
"learning_rate": 3.209366678544267e-05,
"loss": 3.539649658203125,
"step": 8600
},
{
"epoch": 7.578397212543554,
"grad_norm": 2.042051315307617,
"learning_rate": 3.002140150364511e-05,
"loss": 3.5744259643554686,
"step": 8700
},
{
"epoch": 7.665505226480836,
"grad_norm": 2.2132978439331055,
"learning_rate": 2.800641987119348e-05,
"loss": 3.5455661010742188,
"step": 8800
},
{
"epoch": 7.7526132404181185,
"grad_norm": 2.3339362144470215,
"learning_rate": 2.6050371328113e-05,
"loss": 3.5393295288085938,
"step": 8900
},
{
"epoch": 7.839721254355401,
"grad_norm": 2.091668128967285,
"learning_rate": 2.4154857072502135e-05,
"loss": 3.5512890625,
"step": 9000
},
{
"epoch": 7.926829268292683,
"grad_norm": 1.863627552986145,
"learning_rate": 2.232142874981077e-05,
"loss": 3.554620361328125,
"step": 9100
},
{
"epoch": 8.013937282229966,
"grad_norm": 2.4418959617614746,
"learning_rate": 2.055158718268182e-05,
"loss": 3.52468994140625,
"step": 9200
},
{
"epoch": 8.101045296167248,
"grad_norm": 2.270392417907715,
"learning_rate": 1.884678114239542e-05,
"loss": 3.488939208984375,
"step": 9300
},
{
"epoch": 8.18815331010453,
"grad_norm": 1.8760238885879517,
"learning_rate": 1.7208406162922185e-05,
"loss": 3.480302734375,
"step": 9400
},
{
"epoch": 8.275261324041812,
"grad_norm": 1.9609935283660889,
"learning_rate": 1.5637803398555462e-05,
"loss": 3.4763052368164065,
"step": 9500
},
{
"epoch": 8.362369337979095,
"grad_norm": 1.92599356174469,
"learning_rate": 1.4136258526058632e-05,
"loss": 3.4983941650390626,
"step": 9600
},
{
"epoch": 8.449477351916377,
"grad_norm": 1.9898569583892822,
"learning_rate": 1.2705000692225133e-05,
"loss": 3.48035888671875,
"step": 9700
},
{
"epoch": 8.536585365853659,
"grad_norm": 1.9657682180404663,
"learning_rate": 1.1345201507713633e-05,
"loss": 3.4602105712890623,
"step": 9800
},
{
"epoch": 8.623693379790941,
"grad_norm": 2.095853567123413,
"learning_rate": 1.0057974087981414e-05,
"loss": 3.4798574829101563,
"step": 9900
},
{
"epoch": 8.710801393728223,
"grad_norm": 1.7846027612686157,
"learning_rate": 8.844372142101364e-06,
"loss": 3.4902532958984374,
"step": 10000
},
{
"epoch": 8.797909407665506,
"grad_norm": 2.2382795810699463,
"learning_rate": 7.705389110208106e-06,
"loss": 3.4924716186523437,
"step": 10100
},
{
"epoch": 8.885017421602788,
"grad_norm": 1.776693344116211,
"learning_rate": 6.6419573502798374e-06,
"loss": 3.5006637573242188,
"step": 10200
},
{
"epoch": 8.97212543554007,
"grad_norm": 1.9248723983764648,
"learning_rate": 5.65494737492106e-06,
"loss": 3.4578341674804687,
"step": 10300
},
{
"epoch": 9.059233449477352,
"grad_norm": 2.082202911376953,
"learning_rate": 4.7451671387714335e-06,
"loss": 3.461600341796875,
"step": 10400
},
{
"epoch": 9.146341463414634,
"grad_norm": 1.8617357015609741,
"learning_rate": 3.913361377123581e-06,
"loss": 3.477802429199219,
"step": 10500
},
{
"epoch": 9.233449477351916,
"grad_norm": 1.711703896522522,
"learning_rate": 3.1602109962916905e-06,
"loss": 3.442166748046875,
"step": 10600
},
{
"epoch": 9.320557491289199,
"grad_norm": 2.361995220184326,
"learning_rate": 2.4863325162296726e-06,
"loss": 3.454795227050781,
"step": 10700
},
{
"epoch": 9.40766550522648,
"grad_norm": 1.8906856775283813,
"learning_rate": 1.8922775658552715e-06,
"loss": 3.4653359985351564,
"step": 10800
},
{
"epoch": 9.494773519163763,
"grad_norm": 2.1279709339141846,
"learning_rate": 1.3785324314931847e-06,
"loss": 3.4191864013671873,
"step": 10900
},
{
"epoch": 9.581881533101045,
"grad_norm": 2.1340105533599854,
"learning_rate": 9.455176588068382e-07,
"loss": 3.4499856567382814,
"step": 11000
},
{
"epoch": 9.668989547038327,
"grad_norm": 2.4744954109191895,
"learning_rate": 5.935877085446851e-07,
"loss": 3.446976318359375,
"step": 11100
},
{
"epoch": 9.75609756097561,
"grad_norm": 1.9351354837417603,
"learning_rate": 3.230306663828841e-07,
"loss": 3.423928527832031,
"step": 11200
},
{
"epoch": 9.843205574912892,
"grad_norm": 2.3089027404785156,
"learning_rate": 1.3406800710182853e-07,
"loss": 3.4793524169921874,
"step": 11300
},
{
"epoch": 9.930313588850174,
"grad_norm": 2.702860116958618,
"learning_rate": 2.685441328938998e-08,
"loss": 3.4213021850585936,
"step": 11400
}
],
"logging_steps": 100,
"max_steps": 11480,
"num_input_tokens_seen": 0,
"num_train_epochs": 10,
"save_steps": 1000,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 3002513580490752.0,
"train_batch_size": 32,
"trial_name": null,
"trial_params": null
}