vintern-cc4 / trainer_state.json
HUY2612's picture
Upload folder using huggingface_hub
8f5907e verified
Raw
History Blame Contribute Delete
31.5 kB
{
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 3.9851471008283346,
"eval_steps": 500,
"global_step": 872,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.022850614110254214,
"grad_norm": 0.03844932094216347,
"learning_rate": 3.7037037037037037e-05,
"loss": 0.2882,
"step": 5
},
{
"epoch": 0.04570122822050843,
"grad_norm": 0.017711207270622253,
"learning_rate": 7.407407407407407e-05,
"loss": 0.1712,
"step": 10
},
{
"epoch": 0.06855184233076264,
"grad_norm": 0.011037125252187252,
"learning_rate": 0.00011111111111111112,
"loss": 0.0726,
"step": 15
},
{
"epoch": 0.09140245644101685,
"grad_norm": 0.01301049068570137,
"learning_rate": 0.00014814814814814815,
"loss": 0.0664,
"step": 20
},
{
"epoch": 0.11425307055127107,
"grad_norm": 0.009315604344010353,
"learning_rate": 0.0001851851851851852,
"loss": 0.0647,
"step": 25
},
{
"epoch": 0.13710368466152528,
"grad_norm": 0.010555571876466274,
"learning_rate": 0.00019999377994336603,
"loss": 0.0541,
"step": 30
},
{
"epoch": 0.1599542987717795,
"grad_norm": 0.009893398731946945,
"learning_rate": 0.000199955771288296,
"loss": 0.0434,
"step": 35
},
{
"epoch": 0.1828049128820337,
"grad_norm": 0.0062404111959040165,
"learning_rate": 0.00019988322268323268,
"loss": 0.0557,
"step": 40
},
{
"epoch": 0.20565552699228792,
"grad_norm": 0.011121349409222603,
"learning_rate": 0.00019977615919751572,
"loss": 0.0491,
"step": 45
},
{
"epoch": 0.22850614110254214,
"grad_norm": 0.010451147332787514,
"learning_rate": 0.00019963461782718245,
"loss": 0.0428,
"step": 50
},
{
"epoch": 0.2513567552127963,
"grad_norm": 0.008701572194695473,
"learning_rate": 0.0001994586474821837,
"loss": 0.0446,
"step": 55
},
{
"epoch": 0.27420736932305056,
"grad_norm": 0.006607332266867161,
"learning_rate": 0.0001992483089694827,
"loss": 0.0516,
"step": 60
},
{
"epoch": 0.29705798343330475,
"grad_norm": 0.00916219875216484,
"learning_rate": 0.0001990036749720433,
"loss": 0.0378,
"step": 65
},
{
"epoch": 0.319908597543559,
"grad_norm": 0.011637752875685692,
"learning_rate": 0.00019872483002371412,
"loss": 0.0556,
"step": 70
},
{
"epoch": 0.3427592116538132,
"grad_norm": 0.012677548453211784,
"learning_rate": 0.00019841187048001772,
"loss": 0.0376,
"step": 75
},
{
"epoch": 0.3656098257640674,
"grad_norm": 0.007504571694880724,
"learning_rate": 0.00019806490448485463,
"loss": 0.0555,
"step": 80
},
{
"epoch": 0.3884604398743216,
"grad_norm": 0.008339454419910908,
"learning_rate": 0.000197684051933134,
"loss": 0.0412,
"step": 85
},
{
"epoch": 0.41131105398457585,
"grad_norm": 0.006037685554474592,
"learning_rate": 0.00019726944442934377,
"loss": 0.0341,
"step": 90
},
{
"epoch": 0.43416166809483003,
"grad_norm": 0.009660422801971436,
"learning_rate": 0.00019682122524207425,
"loss": 0.0402,
"step": 95
},
{
"epoch": 0.4570122822050843,
"grad_norm": 0.0047590904869139194,
"learning_rate": 0.00019633954925451138,
"loss": 0.0324,
"step": 100
},
{
"epoch": 0.47986289631533846,
"grad_norm": 0.007860913872718811,
"learning_rate": 0.00019582458291091663,
"loss": 0.038,
"step": 105
},
{
"epoch": 0.5027135104255926,
"grad_norm": 0.010165887884795666,
"learning_rate": 0.00019527650415911154,
"loss": 0.051,
"step": 110
},
{
"epoch": 0.5255641245358469,
"grad_norm": 0.010631492361426353,
"learning_rate": 0.00019469550238898758,
"loss": 0.035,
"step": 115
},
{
"epoch": 0.5484147386461011,
"grad_norm": 0.010700591839849949,
"learning_rate": 0.00019408177836706213,
"loss": 0.0496,
"step": 120
},
{
"epoch": 0.5712653527563554,
"grad_norm": 0.005279723089188337,
"learning_rate": 0.00019343554416710292,
"loss": 0.0321,
"step": 125
},
{
"epoch": 0.5941159668666095,
"grad_norm": 0.006138530559837818,
"learning_rate": 0.0001927570230968456,
"loss": 0.0444,
"step": 130
},
{
"epoch": 0.6169665809768637,
"grad_norm": 0.006301143206655979,
"learning_rate": 0.00019204644962082915,
"loss": 0.0366,
"step": 135
},
{
"epoch": 0.639817195087118,
"grad_norm": 0.023399384692311287,
"learning_rate": 0.00019130406927937615,
"loss": 0.0592,
"step": 140
},
{
"epoch": 0.6626678091973722,
"grad_norm": 0.006289128679782152,
"learning_rate": 0.00019053013860374587,
"loss": 0.041,
"step": 145
},
{
"epoch": 0.6855184233076264,
"grad_norm": 0.01251731812953949,
"learning_rate": 0.000189724925027489,
"loss": 0.0555,
"step": 150
},
{
"epoch": 0.7083690374178806,
"grad_norm": 0.007725547533482313,
"learning_rate": 0.0001888887067940356,
"loss": 0.0262,
"step": 155
},
{
"epoch": 0.7312196515281348,
"grad_norm": 0.0047344183549284935,
"learning_rate": 0.0001880217728605474,
"loss": 0.0597,
"step": 160
},
{
"epoch": 0.7540702656383891,
"grad_norm": 0.002183922566473484,
"learning_rate": 0.00018712442279806787,
"loss": 0.0442,
"step": 165
},
{
"epoch": 0.7769208797486432,
"grad_norm": 0.012528673745691776,
"learning_rate": 0.00018619696668800492,
"loss": 0.0386,
"step": 170
},
{
"epoch": 0.7997714938588975,
"grad_norm": 0.010934452526271343,
"learning_rate": 0.00018523972501498137,
"loss": 0.0605,
"step": 175
},
{
"epoch": 0.8226221079691517,
"grad_norm": 0.006062965840101242,
"learning_rate": 0.00018425302855609087,
"loss": 0.0264,
"step": 180
},
{
"epoch": 0.8454727220794059,
"grad_norm": 0.006135896313935518,
"learning_rate": 0.00018323721826659698,
"loss": 0.0515,
"step": 185
},
{
"epoch": 0.8683233361896601,
"grad_norm": 0.010229749605059624,
"learning_rate": 0.00018219264516211543,
"loss": 0.0406,
"step": 190
},
{
"epoch": 0.8911739502999143,
"grad_norm": 0.00646389601752162,
"learning_rate": 0.00018111967019731977,
"loss": 0.0409,
"step": 195
},
{
"epoch": 0.9140245644101685,
"grad_norm": 0.006931021809577942,
"learning_rate": 0.0001800186641412126,
"loss": 0.0344,
"step": 200
},
{
"epoch": 0.9368751785204228,
"grad_norm": 0.0071966927498579025,
"learning_rate": 0.0001788900074490056,
"loss": 0.0366,
"step": 205
},
{
"epoch": 0.9597257926306769,
"grad_norm": 0.014032403007149696,
"learning_rate": 0.0001777340901306522,
"loss": 0.0407,
"step": 210
},
{
"epoch": 0.9825764067409312,
"grad_norm": 0.006412023678421974,
"learning_rate": 0.00017655131161607886,
"loss": 0.0437,
"step": 215
},
{
"epoch": 1.0054270208511853,
"grad_norm": 0.002949735149741173,
"learning_rate": 0.000175342080617161,
"loss": 0.035,
"step": 220
},
{
"epoch": 1.0282776349614395,
"grad_norm": 0.008895672857761383,
"learning_rate": 0.00017410681498649183,
"loss": 0.0168,
"step": 225
},
{
"epoch": 1.0511282490716938,
"grad_norm": 0.012618395499885082,
"learning_rate": 0.00017284594157299218,
"loss": 0.0257,
"step": 230
},
{
"epoch": 1.073978863181948,
"grad_norm": 0.00921704713255167,
"learning_rate": 0.00017155989607441213,
"loss": 0.0286,
"step": 235
},
{
"epoch": 1.0968294772922023,
"grad_norm": 0.005029382184147835,
"learning_rate": 0.00017024912288677435,
"loss": 0.0249,
"step": 240
},
{
"epoch": 1.1196800914024565,
"grad_norm": 0.005939665250480175,
"learning_rate": 0.00016891407495081226,
"loss": 0.0244,
"step": 245
},
{
"epoch": 1.1425307055127107,
"grad_norm": 0.009432048536837101,
"learning_rate": 0.00016755521359545518,
"loss": 0.0206,
"step": 250
},
{
"epoch": 1.165381319622965,
"grad_norm": 0.004679668229073286,
"learning_rate": 0.00016617300837841502,
"loss": 0.0269,
"step": 255
},
{
"epoch": 1.188231933733219,
"grad_norm": 0.0064927395433187485,
"learning_rate": 0.00016476793692392965,
"loss": 0.0205,
"step": 260
},
{
"epoch": 1.2110825478434732,
"grad_norm": 0.0039772894233465195,
"learning_rate": 0.00016334048475771855,
"loss": 0.0183,
"step": 265
},
{
"epoch": 1.2339331619537275,
"grad_norm": 0.007143892347812653,
"learning_rate": 0.00016189114513920838,
"loss": 0.0262,
"step": 270
},
{
"epoch": 1.2567837760639817,
"grad_norm": 0.00343668763525784,
"learning_rate": 0.0001604204188910861,
"loss": 0.014,
"step": 275
},
{
"epoch": 1.279634390174236,
"grad_norm": 0.0038999791722744703,
"learning_rate": 0.00015892881422623827,
"loss": 0.0183,
"step": 280
},
{
"epoch": 1.3024850042844902,
"grad_norm": 0.003784347791224718,
"learning_rate": 0.00015741684657213725,
"loss": 0.0406,
"step": 285
},
{
"epoch": 1.3253356183947442,
"grad_norm": 0.009119242429733276,
"learning_rate": 0.0001558850383927337,
"loss": 0.0205,
"step": 290
},
{
"epoch": 1.3481862325049985,
"grad_norm": 0.006821426562964916,
"learning_rate": 0.00015433391900791817,
"loss": 0.0237,
"step": 295
},
{
"epoch": 1.3710368466152527,
"grad_norm": 0.008907237090170383,
"learning_rate": 0.0001527640244106133,
"loss": 0.0193,
"step": 300
},
{
"epoch": 1.393887460725507,
"grad_norm": 0.009506676346063614,
"learning_rate": 0.00015117589708156013,
"loss": 0.0214,
"step": 305
},
{
"epoch": 1.4167380748357612,
"grad_norm": 0.004221646580845118,
"learning_rate": 0.00014957008580186276,
"loss": 0.0136,
"step": 310
},
{
"epoch": 1.4395886889460154,
"grad_norm": 0.005160444416105747,
"learning_rate": 0.00014794714546335577,
"loss": 0.0282,
"step": 315
},
{
"epoch": 1.4624393030562697,
"grad_norm": 0.008521094918251038,
"learning_rate": 0.00014630763687685988,
"loss": 0.0242,
"step": 320
},
{
"epoch": 1.485289917166524,
"grad_norm": 0.005194018129259348,
"learning_rate": 0.00014465212657839254,
"loss": 0.0146,
"step": 325
},
{
"epoch": 1.5081405312767782,
"grad_norm": 0.005989014636725187,
"learning_rate": 0.00014298118663340017,
"loss": 0.0237,
"step": 330
},
{
"epoch": 1.5309911453870324,
"grad_norm": 0.0050842612981796265,
"learning_rate": 0.0001412953944390795,
"loss": 0.0271,
"step": 335
},
{
"epoch": 1.5538417594972866,
"grad_norm": 0.004291810095310211,
"learning_rate": 0.00013959533252485678,
"loss": 0.0284,
"step": 340
},
{
"epoch": 1.5766923736075407,
"grad_norm": 0.009329627268016338,
"learning_rate": 0.00013788158835109312,
"loss": 0.0273,
"step": 345
},
{
"epoch": 1.599542987717795,
"grad_norm": 0.005101538263261318,
"learning_rate": 0.00013615475410608648,
"loss": 0.025,
"step": 350
},
{
"epoch": 1.6223936018280491,
"grad_norm": 0.004806031938642263,
"learning_rate": 0.0001344154265014393,
"loss": 0.0223,
"step": 355
},
{
"epoch": 1.6452442159383034,
"grad_norm": 0.005375508219003677,
"learning_rate": 0.0001326642065658638,
"loss": 0.0187,
"step": 360
},
{
"epoch": 1.6680948300485574,
"grad_norm": 0.004459770396351814,
"learning_rate": 0.00013090169943749476,
"loss": 0.0287,
"step": 365
},
{
"epoch": 1.6909454441588117,
"grad_norm": 0.004444418475031853,
"learning_rate": 0.0001291285141547828,
"loss": 0.0157,
"step": 370
},
{
"epoch": 1.713796058269066,
"grad_norm": 0.001563582569360733,
"learning_rate": 0.0001273452634460397,
"loss": 0.0137,
"step": 375
},
{
"epoch": 1.7366466723793201,
"grad_norm": 0.012105684727430344,
"learning_rate": 0.00012555256351770873,
"loss": 0.0404,
"step": 380
},
{
"epoch": 1.7594972864895744,
"grad_norm": 0.0045092240907251835,
"learning_rate": 0.00012375103384143295,
"loss": 0.0252,
"step": 385
},
{
"epoch": 1.7823479005998286,
"grad_norm": 0.005620603449642658,
"learning_rate": 0.00012194129693999548,
"loss": 0.0346,
"step": 390
},
{
"epoch": 1.8051985147100829,
"grad_norm": 0.00406628055498004,
"learning_rate": 0.00012012397817220522,
"loss": 0.0155,
"step": 395
},
{
"epoch": 1.828049128820337,
"grad_norm": 0.004286411218345165,
"learning_rate": 0.0001182997055168027,
"loss": 0.0223,
"step": 400
},
{
"epoch": 1.8508997429305913,
"grad_norm": 0.012442640960216522,
"learning_rate": 0.00011646910935546053,
"loss": 0.0196,
"step": 405
},
{
"epoch": 1.8737503570408456,
"grad_norm": 0.0035216982942074537,
"learning_rate": 0.00011463282225495357,
"loss": 0.0247,
"step": 410
},
{
"epoch": 1.8966009711510998,
"grad_norm": 0.004480506759136915,
"learning_rate": 0.00011279147874857396,
"loss": 0.0166,
"step": 415
},
{
"epoch": 1.919451585261354,
"grad_norm": 0.004787661135196686,
"learning_rate": 0.00011094571511686669,
"loss": 0.0322,
"step": 420
},
{
"epoch": 1.942302199371608,
"grad_norm": 0.004430464934557676,
"learning_rate": 0.00010909616916776137,
"loss": 0.0203,
"step": 425
},
{
"epoch": 1.9651528134818623,
"grad_norm": 0.003938391339033842,
"learning_rate": 0.00010724348001617625,
"loss": 0.0258,
"step": 430
},
{
"epoch": 1.9880034275921166,
"grad_norm": 0.005292808637022972,
"learning_rate": 0.00010538828786317045,
"loss": 0.0223,
"step": 435
},
{
"epoch": 2.0108540417023706,
"grad_norm": 0.003457833779975772,
"learning_rate": 0.0001035312337747212,
"loss": 0.0157,
"step": 440
},
{
"epoch": 2.033704655812625,
"grad_norm": 0.0032057061325758696,
"learning_rate": 0.0001016729594602017,
"loss": 0.0136,
"step": 445
},
{
"epoch": 2.056555269922879,
"grad_norm": 0.00509569002315402,
"learning_rate": 9.981410705063728e-05,
"loss": 0.0139,
"step": 450
},
{
"epoch": 2.0794058840331333,
"grad_norm": 0.00236800336278975,
"learning_rate": 9.795531887681523e-05,
"loss": 0.0091,
"step": 455
},
{
"epoch": 2.1022564981433876,
"grad_norm": 0.0026487780269235373,
"learning_rate": 9.609723724732611e-05,
"loss": 0.0142,
"step": 460
},
{
"epoch": 2.125107112253642,
"grad_norm": 0.0039559826254844666,
"learning_rate": 9.424050422661243e-05,
"loss": 0.0115,
"step": 465
},
{
"epoch": 2.147957726363896,
"grad_norm": 0.00275582168251276,
"learning_rate": 9.238576141310172e-05,
"loss": 0.0132,
"step": 470
},
{
"epoch": 2.1708083404741503,
"grad_norm": 0.0073517621494829655,
"learning_rate": 9.053364971750086e-05,
"loss": 0.0137,
"step": 475
},
{
"epoch": 2.1936589545844045,
"grad_norm": 0.003323187353089452,
"learning_rate": 8.868480914132776e-05,
"loss": 0.012,
"step": 480
},
{
"epoch": 2.2165095686946588,
"grad_norm": 0.004128726199269295,
"learning_rate": 8.683987855575741e-05,
"loss": 0.0124,
"step": 485
},
{
"epoch": 2.239360182804913,
"grad_norm": 0.003930236212909222,
"learning_rate": 8.49994954808583e-05,
"loss": 0.0151,
"step": 490
},
{
"epoch": 2.2622107969151672,
"grad_norm": 0.0041837310418486595,
"learning_rate": 8.316429586529615e-05,
"loss": 0.0145,
"step": 495
},
{
"epoch": 2.2850614110254215,
"grad_norm": 0.0032771469559520483,
"learning_rate": 8.133491386658015e-05,
"loss": 0.0135,
"step": 500
},
{
"epoch": 2.3079120251356757,
"grad_norm": 0.0037709022872149944,
"learning_rate": 7.951198163192839e-05,
"loss": 0.0189,
"step": 505
},
{
"epoch": 2.33076263924593,
"grad_norm": 0.005130920093506575,
"learning_rate": 7.769612907982797e-05,
"loss": 0.0197,
"step": 510
},
{
"epoch": 2.3536132533561838,
"grad_norm": 0.0022313841618597507,
"learning_rate": 7.588798368236518e-05,
"loss": 0.0136,
"step": 515
},
{
"epoch": 2.376463867466438,
"grad_norm": 0.003234229050576687,
"learning_rate": 7.408817024840103e-05,
"loss": 0.0189,
"step": 520
},
{
"epoch": 2.3993144815766922,
"grad_norm": 0.003838989417999983,
"learning_rate": 7.229731070766704e-05,
"loss": 0.015,
"step": 525
},
{
"epoch": 2.4221650956869465,
"grad_norm": 0.002571391174569726,
"learning_rate": 7.051602389585609e-05,
"loss": 0.0084,
"step": 530
},
{
"epoch": 2.4450157097972007,
"grad_norm": 0.0035111745819449425,
"learning_rate": 6.87449253407823e-05,
"loss": 0.0105,
"step": 535
},
{
"epoch": 2.467866323907455,
"grad_norm": 0.0015012812800705433,
"learning_rate": 6.69846270496838e-05,
"loss": 0.0116,
"step": 540
},
{
"epoch": 2.490716938017709,
"grad_norm": 0.004130535759031773,
"learning_rate": 6.523573729774234e-05,
"loss": 0.0106,
"step": 545
},
{
"epoch": 2.5135675521279635,
"grad_norm": 0.0008271544356830418,
"learning_rate": 6.349886041789236e-05,
"loss": 0.0067,
"step": 550
},
{
"epoch": 2.5364181662382177,
"grad_norm": 0.003042809432372451,
"learning_rate": 6.177459659199236e-05,
"loss": 0.0102,
"step": 555
},
{
"epoch": 2.559268780348472,
"grad_norm": 0.00416160561144352,
"learning_rate": 6.006354164343046e-05,
"loss": 0.0177,
"step": 560
},
{
"epoch": 2.582119394458726,
"grad_norm": 0.005738898646086454,
"learning_rate": 5.8366286831236595e-05,
"loss": 0.0111,
"step": 565
},
{
"epoch": 2.6049700085689804,
"grad_norm": 0.0028269574977457523,
"learning_rate": 5.668341864577125e-05,
"loss": 0.0087,
"step": 570
},
{
"epoch": 2.6278206226792347,
"grad_norm": 0.0025306132156401873,
"learning_rate": 5.5015518606062514e-05,
"loss": 0.0101,
"step": 575
},
{
"epoch": 2.6506712367894885,
"grad_norm": 0.003558875061571598,
"learning_rate": 5.336316305886078e-05,
"loss": 0.0071,
"step": 580
},
{
"epoch": 2.6735218508997427,
"grad_norm": 0.004009328316897154,
"learning_rate": 5.172692297948081e-05,
"loss": 0.0104,
"step": 585
},
{
"epoch": 2.696372465009997,
"grad_norm": 0.001938402303494513,
"learning_rate": 5.010736377449983e-05,
"loss": 0.0174,
"step": 590
},
{
"epoch": 2.719223079120251,
"grad_norm": 0.003774677636101842,
"learning_rate": 4.850504508638004e-05,
"loss": 0.0107,
"step": 595
},
{
"epoch": 2.7420736932305054,
"grad_norm": 0.002222988521680236,
"learning_rate": 4.692052060008271e-05,
"loss": 0.0117,
"step": 600
},
{
"epoch": 2.7649243073407597,
"grad_norm": 0.005638027563691139,
"learning_rate": 4.535433785174123e-05,
"loss": 0.019,
"step": 605
},
{
"epoch": 2.787774921451014,
"grad_norm": 0.002720755524933338,
"learning_rate": 4.380703803945869e-05,
"loss": 0.0112,
"step": 610
},
{
"epoch": 2.810625535561268,
"grad_norm": 0.007765288930386305,
"learning_rate": 4.2279155836295426e-05,
"loss": 0.0239,
"step": 615
},
{
"epoch": 2.8334761496715224,
"grad_norm": 0.001852337270975113,
"learning_rate": 4.0771219205511754e-05,
"loss": 0.0081,
"step": 620
},
{
"epoch": 2.8563267637817766,
"grad_norm": 0.003439652966335416,
"learning_rate": 3.9283749218128885e-05,
"loss": 0.019,
"step": 625
},
{
"epoch": 2.879177377892031,
"grad_norm": 0.000924278749153018,
"learning_rate": 3.7817259872871656e-05,
"loss": 0.007,
"step": 630
},
{
"epoch": 2.902027992002285,
"grad_norm": 0.0024894895032048225,
"learning_rate": 3.6372257918555e-05,
"loss": 0.0089,
"step": 635
},
{
"epoch": 2.9248786061125394,
"grad_norm": 0.001816246658563614,
"learning_rate": 3.4949242678975846e-05,
"loss": 0.0103,
"step": 640
},
{
"epoch": 2.9477292202227936,
"grad_norm": 0.002363559091463685,
"learning_rate": 3.354870588037054e-05,
"loss": 0.0129,
"step": 645
},
{
"epoch": 2.970579834333048,
"grad_norm": 0.0028436542488634586,
"learning_rate": 3.217113148149765e-05,
"loss": 0.015,
"step": 650
},
{
"epoch": 2.993430448443302,
"grad_norm": 0.0017001379746943712,
"learning_rate": 3.0816995506404996e-05,
"loss": 0.0113,
"step": 655
},
{
"epoch": 3.0162810625535563,
"grad_norm": 0.004137461539357901,
"learning_rate": 2.948676587993834e-05,
"loss": 0.0066,
"step": 660
},
{
"epoch": 3.03913167666381,
"grad_norm": 0.004200291819870472,
"learning_rate": 2.8180902266048948e-05,
"loss": 0.0078,
"step": 665
},
{
"epoch": 3.0619822907740644,
"grad_norm": 0.0025121369399130344,
"learning_rate": 2.6899855908955464e-05,
"loss": 0.0085,
"step": 670
},
{
"epoch": 3.0848329048843186,
"grad_norm": 0.0015045857289806008,
"learning_rate": 2.564406947721566e-05,
"loss": 0.0073,
"step": 675
},
{
"epoch": 3.107683518994573,
"grad_norm": 0.002549678785726428,
"learning_rate": 2.4413976910761116e-05,
"loss": 0.0104,
"step": 680
},
{
"epoch": 3.130534133104827,
"grad_norm": 0.0011935221264138818,
"learning_rate": 2.3210003270948365e-05,
"loss": 0.0114,
"step": 685
},
{
"epoch": 3.1533847472150813,
"grad_norm": 0.000922174658626318,
"learning_rate": 2.2032564593677774e-05,
"loss": 0.0094,
"step": 690
},
{
"epoch": 3.1762353613253356,
"grad_norm": 0.0038747189100831747,
"learning_rate": 2.0882067745631605e-05,
"loss": 0.0092,
"step": 695
},
{
"epoch": 3.19908597543559,
"grad_norm": 0.004468169994652271,
"learning_rate": 1.9758910283680132e-05,
"loss": 0.007,
"step": 700
},
{
"epoch": 3.221936589545844,
"grad_norm": 0.0032682984601706266,
"learning_rate": 1.8663480317504988e-05,
"loss": 0.0102,
"step": 705
},
{
"epoch": 3.2447872036560983,
"grad_norm": 0.0015670544235035777,
"learning_rate": 1.7596156375486862e-05,
"loss": 0.0096,
"step": 710
},
{
"epoch": 3.2676378177663525,
"grad_norm": 0.003037898801267147,
"learning_rate": 1.6557307273904354e-05,
"loss": 0.0106,
"step": 715
},
{
"epoch": 3.290488431876607,
"grad_norm": 0.001072474056854844,
"learning_rate": 1.5547291989488444e-05,
"loss": 0.0065,
"step": 720
},
{
"epoch": 3.313339045986861,
"grad_norm": 0.0019489972619339824,
"learning_rate": 1.4566459535377252e-05,
"loss": 0.005,
"step": 725
},
{
"epoch": 3.3361896600971153,
"grad_norm": 0.0007796495920047164,
"learning_rate": 1.3615148840513881e-05,
"loss": 0.0071,
"step": 730
},
{
"epoch": 3.3590402742073695,
"grad_norm": 0.0009693990577943623,
"learning_rate": 1.2693688632528622e-05,
"loss": 0.0056,
"step": 735
},
{
"epoch": 3.3818908883176233,
"grad_norm": 0.0008724929648451507,
"learning_rate": 1.1802397324146374e-05,
"loss": 0.0069,
"step": 740
},
{
"epoch": 3.4047415024278775,
"grad_norm": 0.001148922834545374,
"learning_rate": 1.0941582903158343e-05,
"loss": 0.0061,
"step": 745
},
{
"epoch": 3.427592116538132,
"grad_norm": 0.0015940605662763119,
"learning_rate": 1.0111542825996245e-05,
"loss": 0.0083,
"step": 750
},
{
"epoch": 3.450442730648386,
"grad_norm": 0.0023965986911207438,
"learning_rate": 9.31256391494546e-06,
"loss": 0.0062,
"step": 755
},
{
"epoch": 3.4732933447586403,
"grad_norm": 0.0008359673665836453,
"learning_rate": 8.54492225903295e-06,
"loss": 0.0057,
"step": 760
},
{
"epoch": 3.4961439588688945,
"grad_norm": 0.0041641597636044025,
"learning_rate": 7.80888311862401e-06,
"loss": 0.0076,
"step": 765
},
{
"epoch": 3.5189945729791487,
"grad_norm": 0.003335606772452593,
"learning_rate": 7.104700833761013e-06,
"loss": 0.0082,
"step": 770
},
{
"epoch": 3.541845187089403,
"grad_norm": 0.0013379153097048402,
"learning_rate": 6.432618736275553e-06,
"loss": 0.0061,
"step": 775
},
{
"epoch": 3.5646958011996572,
"grad_norm": 0.001888817292638123,
"learning_rate": 5.7928690657045535e-06,
"loss": 0.0086,
"step": 780
},
{
"epoch": 3.5875464153099115,
"grad_norm": 0.0022849293891340494,
"learning_rate": 5.185672889039394e-06,
"loss": 0.0081,
"step": 785
},
{
"epoch": 3.6103970294201657,
"grad_norm": 0.001410789554938674,
"learning_rate": 4.611240024335706e-06,
"loss": 0.0082,
"step": 790
},
{
"epoch": 3.63324764353042,
"grad_norm": 0.0019127277191728354,
"learning_rate": 4.069768968210186e-06,
"loss": 0.0054,
"step": 795
},
{
"epoch": 3.656098257640674,
"grad_norm": 0.0023042806424200535,
"learning_rate": 3.561446827249659e-06,
"loss": 0.0116,
"step": 800
},
{
"epoch": 3.6789488717509284,
"grad_norm": 0.002472461434081197,
"learning_rate": 3.0864492533560165e-06,
"loss": 0.0091,
"step": 805
},
{
"epoch": 3.7017994858611827,
"grad_norm": 0.0036690961569547653,
"learning_rate": 2.6449403830492104e-06,
"loss": 0.008,
"step": 810
},
{
"epoch": 3.724650099971437,
"grad_norm": 0.001905747689306736,
"learning_rate": 2.2370727807495497e-06,
"loss": 0.0103,
"step": 815
},
{
"epoch": 3.747500714081691,
"grad_norm": 0.0008393717580474913,
"learning_rate": 1.8629873860586566e-06,
"loss": 0.0041,
"step": 820
},
{
"epoch": 3.7703513281919454,
"grad_norm": 0.0025834240950644016,
"learning_rate": 1.5228134650575265e-06,
"loss": 0.0075,
"step": 825
},
{
"epoch": 3.7932019423021996,
"grad_norm": 0.0007983591058291495,
"learning_rate": 1.2166685656382903e-06,
"loss": 0.0088,
"step": 830
},
{
"epoch": 3.8160525564124534,
"grad_norm": 0.0027825688011944294,
"learning_rate": 9.446584768852407e-07,
"loss": 0.0097,
"step": 835
},
{
"epoch": 3.8389031705227077,
"grad_norm": 0.0036266965325921774,
"learning_rate": 7.068771925192286e-07,
"loss": 0.0073,
"step": 840
},
{
"epoch": 3.861753784632962,
"grad_norm": 0.00241619604639709,
"learning_rate": 5.034068784178891e-07,
"loss": 0.007,
"step": 845
},
{
"epoch": 3.884604398743216,
"grad_norm": 0.0021260769572108984,
"learning_rate": 3.343178442230088e-07,
"loss": 0.0085,
"step": 850
},
{
"epoch": 3.9074550128534704,
"grad_norm": 0.0020669158548116684,
"learning_rate": 1.9966851904487106e-07,
"loss": 0.0081,
"step": 855
},
{
"epoch": 3.9303056269637247,
"grad_norm": 0.0035362038761377335,
"learning_rate": 9.950543127198453e-08,
"loss": 0.0081,
"step": 860
},
{
"epoch": 3.953156241073979,
"grad_norm": 0.001472371513955295,
"learning_rate": 3.386319249303327e-08,
"loss": 0.0062,
"step": 865
},
{
"epoch": 3.976006855184233,
"grad_norm": 0.0012045196490362287,
"learning_rate": 2.764485536776995e-09,
"loss": 0.0075,
"step": 870
},
{
"epoch": 3.9851471008283346,
"step": 872,
"total_flos": 1.3168379251306697e+21,
"train_loss": 0.024513581349010313,
"train_runtime": 198435.7406,
"train_samples_per_second": 0.282,
"train_steps_per_second": 0.004
}
],
"logging_steps": 5,
"max_steps": 872,
"num_input_tokens_seen": 0,
"num_train_epochs": 4,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 1.3168379251306697e+21,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}