CodeIsAbstract's picture
Upload folder using huggingface_hub
fee8119 verified
Raw
History Blame Contribute Delete
31.1 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 929,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.005382131324004306,
"grad_norm": 19.25,
"learning_rate": 5.7142857142857145e-06,
"loss": 7.5826,
"step": 5
},
{
"epoch": 0.010764262648008612,
"grad_norm": 24.125,
"learning_rate": 1.2857142857142857e-05,
"loss": 7.7389,
"step": 10
},
{
"epoch": 0.016146393972012917,
"grad_norm": 20.375,
"learning_rate": 2e-05,
"loss": 8.0474,
"step": 15
},
{
"epoch": 0.021528525296017224,
"grad_norm": 17.375,
"learning_rate": 2.714285714285714e-05,
"loss": 7.2989,
"step": 20
},
{
"epoch": 0.02691065662002153,
"grad_norm": 15.0625,
"learning_rate": 3.428571428571429e-05,
"loss": 7.8579,
"step": 25
},
{
"epoch": 0.03229278794402583,
"grad_norm": 12.875,
"learning_rate": 4.1428571428571437e-05,
"loss": 7.3271,
"step": 30
},
{
"epoch": 0.03767491926803014,
"grad_norm": 8.5,
"learning_rate": 4.8571428571428576e-05,
"loss": 7.0945,
"step": 35
},
{
"epoch": 0.04305705059203445,
"grad_norm": 7.46875,
"learning_rate": 5.571428571428572e-05,
"loss": 6.3577,
"step": 40
},
{
"epoch": 0.04843918191603875,
"grad_norm": 7.40625,
"learning_rate": 6.285714285714286e-05,
"loss": 6.2656,
"step": 45
},
{
"epoch": 0.05382131324004306,
"grad_norm": 3.53125,
"learning_rate": 7e-05,
"loss": 6.2146,
"step": 50
},
{
"epoch": 0.059203444564047365,
"grad_norm": 3.671875,
"learning_rate": 7.714285714285715e-05,
"loss": 5.7147,
"step": 55
},
{
"epoch": 0.06458557588805167,
"grad_norm": 2.6875,
"learning_rate": 8.428571428571429e-05,
"loss": 5.2684,
"step": 60
},
{
"epoch": 0.06996770721205597,
"grad_norm": 2.203125,
"learning_rate": 9.142857142857143e-05,
"loss": 4.914,
"step": 65
},
{
"epoch": 0.07534983853606028,
"grad_norm": 2.28125,
"learning_rate": 9.857142857142858e-05,
"loss": 5.5474,
"step": 70
},
{
"epoch": 0.08073196986006459,
"grad_norm": 3.03125,
"learning_rate": 0.00010571428571428572,
"loss": 4.0901,
"step": 75
},
{
"epoch": 0.0861141011840689,
"grad_norm": 2.40625,
"learning_rate": 0.00011285714285714286,
"loss": 4.1259,
"step": 80
},
{
"epoch": 0.09149623250807319,
"grad_norm": 2.328125,
"learning_rate": 0.00012,
"loss": 3.7235,
"step": 85
},
{
"epoch": 0.0968783638320775,
"grad_norm": 2.46875,
"learning_rate": 0.00012714285714285714,
"loss": 3.1657,
"step": 90
},
{
"epoch": 0.10226049515608181,
"grad_norm": 3.171875,
"learning_rate": 0.00013428571428571428,
"loss": 3.4388,
"step": 95
},
{
"epoch": 0.10764262648008611,
"grad_norm": 2.34375,
"learning_rate": 0.00014142857142857145,
"loss": 3.637,
"step": 100
},
{
"epoch": 0.11302475780409042,
"grad_norm": 1.9375,
"learning_rate": 0.00014857142857142857,
"loss": 2.8895,
"step": 105
},
{
"epoch": 0.11840688912809473,
"grad_norm": 2.40625,
"learning_rate": 0.00015571428571428572,
"loss": 3.011,
"step": 110
},
{
"epoch": 0.12378902045209902,
"grad_norm": 1.6796875,
"learning_rate": 0.00016285714285714287,
"loss": 2.5217,
"step": 115
},
{
"epoch": 0.12917115177610333,
"grad_norm": 1.609375,
"learning_rate": 0.00017,
"loss": 2.3684,
"step": 120
},
{
"epoch": 0.13455328310010764,
"grad_norm": 1.3203125,
"learning_rate": 0.00017714285714285713,
"loss": 2.6268,
"step": 125
},
{
"epoch": 0.13993541442411195,
"grad_norm": 1.6953125,
"learning_rate": 0.00018428571428571428,
"loss": 2.2101,
"step": 130
},
{
"epoch": 0.14531754574811626,
"grad_norm": 1.71875,
"learning_rate": 0.00019142857142857145,
"loss": 2.1823,
"step": 135
},
{
"epoch": 0.15069967707212056,
"grad_norm": 0.9375,
"learning_rate": 0.0001985714285714286,
"loss": 1.8939,
"step": 140
},
{
"epoch": 0.15608180839612487,
"grad_norm": 1.3125,
"learning_rate": 0.00019999887311127372,
"loss": 2.6186,
"step": 145
},
{
"epoch": 0.16146393972012918,
"grad_norm": 1.15625,
"learning_rate": 0.0001999942951693512,
"loss": 2.3448,
"step": 150
},
{
"epoch": 0.1668460710441335,
"grad_norm": 1.53125,
"learning_rate": 0.00019998619590477638,
"loss": 2.0453,
"step": 155
},
{
"epoch": 0.1722282023681378,
"grad_norm": 1.1328125,
"learning_rate": 0.00019997457560276666,
"loss": 1.9159,
"step": 160
},
{
"epoch": 0.17761033369214208,
"grad_norm": 1.3359375,
"learning_rate": 0.00019995943467253385,
"loss": 2.2564,
"step": 165
},
{
"epoch": 0.18299246501614638,
"grad_norm": 1.453125,
"learning_rate": 0.00019994077364726925,
"loss": 2.1459,
"step": 170
},
{
"epoch": 0.1883745963401507,
"grad_norm": 1.15625,
"learning_rate": 0.00019991859318412537,
"loss": 2.3652,
"step": 175
},
{
"epoch": 0.193756727664155,
"grad_norm": 1.703125,
"learning_rate": 0.00019989289406419238,
"loss": 2.0883,
"step": 180
},
{
"epoch": 0.1991388589881593,
"grad_norm": 1.046875,
"learning_rate": 0.00019986367719247082,
"loss": 2.0466,
"step": 185
},
{
"epoch": 0.20452099031216361,
"grad_norm": 1.9765625,
"learning_rate": 0.00019983094359783977,
"loss": 2.0306,
"step": 190
},
{
"epoch": 0.20990312163616792,
"grad_norm": 0.93359375,
"learning_rate": 0.00019979469443302046,
"loss": 1.73,
"step": 195
},
{
"epoch": 0.21528525296017223,
"grad_norm": 1.203125,
"learning_rate": 0.0001997549309745357,
"loss": 2.1835,
"step": 200
},
{
"epoch": 0.22066738428417654,
"grad_norm": 0.80859375,
"learning_rate": 0.00019971165462266512,
"loss": 1.8594,
"step": 205
},
{
"epoch": 0.22604951560818085,
"grad_norm": 0.98828125,
"learning_rate": 0.00019966486690139564,
"loss": 1.8338,
"step": 210
},
{
"epoch": 0.23143164693218515,
"grad_norm": 1.15625,
"learning_rate": 0.0001996145694583678,
"loss": 2.0899,
"step": 215
},
{
"epoch": 0.23681377825618946,
"grad_norm": 1.296875,
"learning_rate": 0.000199560764064818,
"loss": 2.0116,
"step": 220
},
{
"epoch": 0.24219590958019377,
"grad_norm": 1.4375,
"learning_rate": 0.00019950345261551583,
"loss": 1.9816,
"step": 225
},
{
"epoch": 0.24757804090419805,
"grad_norm": 0.89453125,
"learning_rate": 0.0001994426371286974,
"loss": 1.4429,
"step": 230
},
{
"epoch": 0.2529601722282024,
"grad_norm": 1.171875,
"learning_rate": 0.00019937831974599447,
"loss": 1.8191,
"step": 235
},
{
"epoch": 0.25834230355220666,
"grad_norm": 1.0625,
"learning_rate": 0.0001993105027323588,
"loss": 2.0092,
"step": 240
},
{
"epoch": 0.263724434876211,
"grad_norm": 0.95703125,
"learning_rate": 0.00019923918847598247,
"loss": 1.6442,
"step": 245
},
{
"epoch": 0.2691065662002153,
"grad_norm": 1.1953125,
"learning_rate": 0.00019916437948821382,
"loss": 2.29,
"step": 250
},
{
"epoch": 0.2744886975242196,
"grad_norm": 1.2734375,
"learning_rate": 0.00019908607840346903,
"loss": 1.7178,
"step": 255
},
{
"epoch": 0.2798708288482239,
"grad_norm": 0.91015625,
"learning_rate": 0.00019900428797913915,
"loss": 1.8539,
"step": 260
},
{
"epoch": 0.2852529601722282,
"grad_norm": 1.21875,
"learning_rate": 0.0001989190110954933,
"loss": 1.9345,
"step": 265
},
{
"epoch": 0.2906350914962325,
"grad_norm": 1.3515625,
"learning_rate": 0.000198830250755577,
"loss": 2.1381,
"step": 270
},
{
"epoch": 0.2960172228202368,
"grad_norm": 1.015625,
"learning_rate": 0.0001987380100851065,
"loss": 2.0466,
"step": 275
},
{
"epoch": 0.3013993541442411,
"grad_norm": 0.953125,
"learning_rate": 0.00019864229233235875,
"loss": 2.1104,
"step": 280
},
{
"epoch": 0.3067814854682454,
"grad_norm": 0.984375,
"learning_rate": 0.000198543100868057,
"loss": 2.0808,
"step": 285
},
{
"epoch": 0.31216361679224974,
"grad_norm": 1.390625,
"learning_rate": 0.00019844043918525195,
"loss": 1.9378,
"step": 290
},
{
"epoch": 0.317545748116254,
"grad_norm": 0.78515625,
"learning_rate": 0.00019833431089919898,
"loss": 1.565,
"step": 295
},
{
"epoch": 0.32292787944025836,
"grad_norm": 1.7109375,
"learning_rate": 0.00019822471974723064,
"loss": 1.8316,
"step": 300
},
{
"epoch": 0.32831001076426264,
"grad_norm": 1.0234375,
"learning_rate": 0.0001981116695886252,
"loss": 1.9357,
"step": 305
},
{
"epoch": 0.333692142088267,
"grad_norm": 0.99609375,
"learning_rate": 0.00019799516440447058,
"loss": 1.7745,
"step": 310
},
{
"epoch": 0.33907427341227125,
"grad_norm": 0.97265625,
"learning_rate": 0.00019787520829752427,
"loss": 1.8947,
"step": 315
},
{
"epoch": 0.3444564047362756,
"grad_norm": 0.8046875,
"learning_rate": 0.00019775180549206888,
"loss": 1.6128,
"step": 320
},
{
"epoch": 0.34983853606027987,
"grad_norm": 0.921875,
"learning_rate": 0.0001976249603337632,
"loss": 1.6003,
"step": 325
},
{
"epoch": 0.35522066738428415,
"grad_norm": 0.7421875,
"learning_rate": 0.00019749467728948943,
"loss": 1.7543,
"step": 330
},
{
"epoch": 0.3606027987082885,
"grad_norm": 1.46875,
"learning_rate": 0.0001973609609471956,
"loss": 1.8543,
"step": 335
},
{
"epoch": 0.36598493003229277,
"grad_norm": 0.953125,
"learning_rate": 0.00019722381601573413,
"loss": 1.7436,
"step": 340
},
{
"epoch": 0.3713670613562971,
"grad_norm": 0.9765625,
"learning_rate": 0.0001970832473246962,
"loss": 2.1039,
"step": 345
},
{
"epoch": 0.3767491926803014,
"grad_norm": 0.76171875,
"learning_rate": 0.0001969392598242413,
"loss": 1.9624,
"step": 350
},
{
"epoch": 0.3821313240043057,
"grad_norm": 0.6796875,
"learning_rate": 0.0001967918585849232,
"loss": 1.4343,
"step": 355
},
{
"epoch": 0.38751345532831,
"grad_norm": 1.2421875,
"learning_rate": 0.00019664104879751123,
"loss": 1.5505,
"step": 360
},
{
"epoch": 0.39289558665231433,
"grad_norm": 0.86328125,
"learning_rate": 0.00019648683577280757,
"loss": 1.6731,
"step": 365
},
{
"epoch": 0.3982777179763186,
"grad_norm": 0.90234375,
"learning_rate": 0.0001963292249414602,
"loss": 1.6294,
"step": 370
},
{
"epoch": 0.40365984930032295,
"grad_norm": 0.80859375,
"learning_rate": 0.0001961682218537717,
"loss": 1.7152,
"step": 375
},
{
"epoch": 0.40904198062432723,
"grad_norm": 1.2890625,
"learning_rate": 0.00019600383217950365,
"loss": 2.0493,
"step": 380
},
{
"epoch": 0.41442411194833156,
"grad_norm": 1.2890625,
"learning_rate": 0.00019583606170767718,
"loss": 1.809,
"step": 385
},
{
"epoch": 0.41980624327233584,
"grad_norm": 0.93359375,
"learning_rate": 0.00019566491634636897,
"loss": 2.0224,
"step": 390
},
{
"epoch": 0.4251883745963401,
"grad_norm": 1.1484375,
"learning_rate": 0.0001954904021225032,
"loss": 1.8455,
"step": 395
},
{
"epoch": 0.43057050592034446,
"grad_norm": 0.83984375,
"learning_rate": 0.0001953125251816394,
"loss": 1.5119,
"step": 400
},
{
"epoch": 0.43595263724434874,
"grad_norm": 0.95703125,
"learning_rate": 0.00019513129178775588,
"loss": 1.6183,
"step": 405
},
{
"epoch": 0.4413347685683531,
"grad_norm": 0.97265625,
"learning_rate": 0.0001949467083230293,
"loss": 1.5994,
"step": 410
},
{
"epoch": 0.44671689989235736,
"grad_norm": 0.92578125,
"learning_rate": 0.0001947587812876099,
"loss": 1.9839,
"step": 415
},
{
"epoch": 0.4520990312163617,
"grad_norm": 0.95703125,
"learning_rate": 0.00019456751729939238,
"loss": 1.7201,
"step": 420
},
{
"epoch": 0.45748116254036597,
"grad_norm": 0.74609375,
"learning_rate": 0.00019437292309378324,
"loss": 1.8024,
"step": 425
},
{
"epoch": 0.4628632938643703,
"grad_norm": 1.609375,
"learning_rate": 0.00019417500552346316,
"loss": 1.421,
"step": 430
},
{
"epoch": 0.4682454251883746,
"grad_norm": 0.98046875,
"learning_rate": 0.000193973771558146,
"loss": 1.5199,
"step": 435
},
{
"epoch": 0.4736275565123789,
"grad_norm": 0.97265625,
"learning_rate": 0.0001937692282843333,
"loss": 1.8052,
"step": 440
},
{
"epoch": 0.4790096878363832,
"grad_norm": 1.078125,
"learning_rate": 0.00019356138290506461,
"loss": 2.0869,
"step": 445
},
{
"epoch": 0.48439181916038754,
"grad_norm": 0.99609375,
"learning_rate": 0.00019335024273966383,
"loss": 1.5393,
"step": 450
},
{
"epoch": 0.4897739504843918,
"grad_norm": 1.171875,
"learning_rate": 0.00019313581522348164,
"loss": 1.6464,
"step": 455
},
{
"epoch": 0.4951560818083961,
"grad_norm": 0.83203125,
"learning_rate": 0.00019291810790763355,
"loss": 1.8133,
"step": 460
},
{
"epoch": 0.5005382131324004,
"grad_norm": 1.453125,
"learning_rate": 0.00019269712845873397,
"loss": 2.0506,
"step": 465
},
{
"epoch": 0.5059203444564048,
"grad_norm": 0.84375,
"learning_rate": 0.00019247288465862615,
"loss": 1.95,
"step": 470
},
{
"epoch": 0.511302475780409,
"grad_norm": 0.91015625,
"learning_rate": 0.0001922453844041084,
"loss": 1.565,
"step": 475
},
{
"epoch": 0.5166846071044133,
"grad_norm": 0.71875,
"learning_rate": 0.00019201463570665573,
"loss": 1.7664,
"step": 480
},
{
"epoch": 0.5220667384284177,
"grad_norm": 0.6015625,
"learning_rate": 0.0001917806466921378,
"loss": 1.6515,
"step": 485
},
{
"epoch": 0.527448869752422,
"grad_norm": 0.69140625,
"learning_rate": 0.00019154342560053297,
"loss": 1.654,
"step": 490
},
{
"epoch": 0.5328310010764262,
"grad_norm": 0.62109375,
"learning_rate": 0.0001913029807856378,
"loss": 1.5976,
"step": 495
},
{
"epoch": 0.5382131324004306,
"grad_norm": 1.0859375,
"learning_rate": 0.00019105932071477303,
"loss": 1.497,
"step": 500
},
{
"epoch": 0.5435952637244349,
"grad_norm": 0.87890625,
"learning_rate": 0.00019081245396848547,
"loss": 1.6065,
"step": 505
},
{
"epoch": 0.5489773950484392,
"grad_norm": 0.78125,
"learning_rate": 0.00019056238924024574,
"loss": 1.3089,
"step": 510
},
{
"epoch": 0.5543595263724435,
"grad_norm": 1.0546875,
"learning_rate": 0.00019030913533614208,
"loss": 1.8274,
"step": 515
},
{
"epoch": 0.5597416576964478,
"grad_norm": 1.1953125,
"learning_rate": 0.00019005270117457043,
"loss": 1.6165,
"step": 520
},
{
"epoch": 0.5651237890204521,
"grad_norm": 1.0625,
"learning_rate": 0.00018979309578592016,
"loss": 1.578,
"step": 525
},
{
"epoch": 0.5705059203444564,
"grad_norm": 1.7109375,
"learning_rate": 0.0001895303283122561,
"loss": 1.7197,
"step": 530
},
{
"epoch": 0.5758880516684607,
"grad_norm": 1.0390625,
"learning_rate": 0.00018926440800699678,
"loss": 1.7235,
"step": 535
},
{
"epoch": 0.581270182992465,
"grad_norm": 0.92578125,
"learning_rate": 0.0001889953442345884,
"loss": 1.536,
"step": 540
},
{
"epoch": 0.5866523143164694,
"grad_norm": 0.8125,
"learning_rate": 0.000188723146470175,
"loss": 1.8374,
"step": 545
},
{
"epoch": 0.5920344456404736,
"grad_norm": 0.96484375,
"learning_rate": 0.00018844782429926495,
"loss": 1.4953,
"step": 550
},
{
"epoch": 0.5974165769644779,
"grad_norm": 0.7734375,
"learning_rate": 0.0001881693874173934,
"loss": 1.3424,
"step": 555
},
{
"epoch": 0.6027987082884823,
"grad_norm": 0.7890625,
"learning_rate": 0.0001878878456297807,
"loss": 1.4368,
"step": 560
},
{
"epoch": 0.6081808396124866,
"grad_norm": 1.09375,
"learning_rate": 0.00018760320885098715,
"loss": 1.7983,
"step": 565
},
{
"epoch": 0.6135629709364908,
"grad_norm": 1.1015625,
"learning_rate": 0.00018731548710456398,
"loss": 1.8513,
"step": 570
},
{
"epoch": 0.6189451022604952,
"grad_norm": 2.125,
"learning_rate": 0.00018702469052270023,
"loss": 2.1254,
"step": 575
},
{
"epoch": 0.6243272335844995,
"grad_norm": 0.7421875,
"learning_rate": 0.00018673082934586607,
"loss": 1.6763,
"step": 580
},
{
"epoch": 0.6297093649085038,
"grad_norm": 1.140625,
"learning_rate": 0.00018643391392245197,
"loss": 1.6641,
"step": 585
},
{
"epoch": 0.635091496232508,
"grad_norm": 0.890625,
"learning_rate": 0.00018613395470840454,
"loss": 2.0576,
"step": 590
},
{
"epoch": 0.6404736275565124,
"grad_norm": 0.74609375,
"learning_rate": 0.0001858309622668581,
"loss": 1.643,
"step": 595
},
{
"epoch": 0.6458557588805167,
"grad_norm": 1.3671875,
"learning_rate": 0.00018552494726776283,
"loss": 1.8816,
"step": 600
},
{
"epoch": 0.6512378902045209,
"grad_norm": 0.8515625,
"learning_rate": 0.00018521592048750907,
"loss": 1.4728,
"step": 605
},
{
"epoch": 0.6566200215285253,
"grad_norm": 0.94921875,
"learning_rate": 0.00018490389280854754,
"loss": 1.6078,
"step": 610
},
{
"epoch": 0.6620021528525296,
"grad_norm": 0.96484375,
"learning_rate": 0.0001845888752190065,
"loss": 1.6509,
"step": 615
},
{
"epoch": 0.667384284176534,
"grad_norm": 0.73046875,
"learning_rate": 0.00018427087881230455,
"loss": 1.1777,
"step": 620
},
{
"epoch": 0.6727664155005382,
"grad_norm": 1.09375,
"learning_rate": 0.00018394991478676005,
"loss": 1.4283,
"step": 625
},
{
"epoch": 0.6781485468245425,
"grad_norm": 0.859375,
"learning_rate": 0.0001836259944451967,
"loss": 1.6932,
"step": 630
},
{
"epoch": 0.6835306781485468,
"grad_norm": 1.03125,
"learning_rate": 0.00018329912919454565,
"loss": 1.4168,
"step": 635
},
{
"epoch": 0.6889128094725512,
"grad_norm": 0.77734375,
"learning_rate": 0.00018296933054544367,
"loss": 1.7149,
"step": 640
},
{
"epoch": 0.6942949407965554,
"grad_norm": 0.921875,
"learning_rate": 0.00018263661011182783,
"loss": 1.5492,
"step": 645
},
{
"epoch": 0.6996770721205597,
"grad_norm": 0.8828125,
"learning_rate": 0.00018230097961052658,
"loss": 1.8432,
"step": 650
},
{
"epoch": 0.7050592034445641,
"grad_norm": 0.8359375,
"learning_rate": 0.00018196245086084703,
"loss": 1.5463,
"step": 655
},
{
"epoch": 0.7104413347685683,
"grad_norm": 1.03125,
"learning_rate": 0.00018162103578415884,
"loss": 1.6883,
"step": 660
},
{
"epoch": 0.7158234660925726,
"grad_norm": 0.8515625,
"learning_rate": 0.0001812767464034743,
"loss": 1.5512,
"step": 665
},
{
"epoch": 0.721205597416577,
"grad_norm": 0.91796875,
"learning_rate": 0.00018092959484302515,
"loss": 1.387,
"step": 670
},
{
"epoch": 0.7265877287405813,
"grad_norm": 0.7421875,
"learning_rate": 0.00018057959332783518,
"loss": 1.2412,
"step": 675
},
{
"epoch": 0.7319698600645855,
"grad_norm": 0.6875,
"learning_rate": 0.0001802267541832903,
"loss": 1.4303,
"step": 680
},
{
"epoch": 0.7373519913885899,
"grad_norm": 1.359375,
"learning_rate": 0.00017987108983470403,
"loss": 1.9346,
"step": 685
},
{
"epoch": 0.7427341227125942,
"grad_norm": 1.890625,
"learning_rate": 0.0001795126128068801,
"loss": 1.5194,
"step": 690
},
{
"epoch": 0.7481162540365985,
"grad_norm": 1.1796875,
"learning_rate": 0.00017915133572367155,
"loss": 1.9612,
"step": 695
},
{
"epoch": 0.7534983853606028,
"grad_norm": 1.15625,
"learning_rate": 0.00017878727130753592,
"loss": 1.8595,
"step": 700
},
{
"epoch": 0.7588805166846071,
"grad_norm": 1.1640625,
"learning_rate": 0.00017842043237908733,
"loss": 1.4777,
"step": 705
},
{
"epoch": 0.7642626480086114,
"grad_norm": 1.1171875,
"learning_rate": 0.00017805083185664508,
"loss": 1.6248,
"step": 710
},
{
"epoch": 0.7696447793326158,
"grad_norm": 0.82421875,
"learning_rate": 0.00017767848275577856,
"loss": 1.7387,
"step": 715
},
{
"epoch": 0.77502691065662,
"grad_norm": 0.83984375,
"learning_rate": 0.0001773033981888491,
"loss": 1.7641,
"step": 720
},
{
"epoch": 0.7804090419806243,
"grad_norm": 0.8359375,
"learning_rate": 0.000176925591364548,
"loss": 1.853,
"step": 725
},
{
"epoch": 0.7857911733046287,
"grad_norm": 1.046875,
"learning_rate": 0.00017654507558743153,
"loss": 1.3802,
"step": 730
},
{
"epoch": 0.7911733046286329,
"grad_norm": 0.87890625,
"learning_rate": 0.00017616186425745248,
"loss": 1.8941,
"step": 735
},
{
"epoch": 0.7965554359526372,
"grad_norm": 0.80078125,
"learning_rate": 0.00017577597086948797,
"loss": 1.6458,
"step": 740
},
{
"epoch": 0.8019375672766416,
"grad_norm": 1.0078125,
"learning_rate": 0.00017538740901286464,
"loss": 1.6792,
"step": 745
},
{
"epoch": 0.8073196986006459,
"grad_norm": 0.640625,
"learning_rate": 0.00017499619237087969,
"loss": 1.389,
"step": 750
},
{
"epoch": 0.8127018299246501,
"grad_norm": 1.015625,
"learning_rate": 0.00017460233472031935,
"loss": 2.0551,
"step": 755
},
{
"epoch": 0.8180839612486545,
"grad_norm": 1.0625,
"learning_rate": 0.0001742058499309735,
"loss": 1.6721,
"step": 760
},
{
"epoch": 0.8234660925726588,
"grad_norm": 0.83203125,
"learning_rate": 0.00017380675196514739,
"loss": 1.3685,
"step": 765
},
{
"epoch": 0.8288482238966631,
"grad_norm": 1.09375,
"learning_rate": 0.00017340505487716985,
"loss": 1.7255,
"step": 770
},
{
"epoch": 0.8342303552206674,
"grad_norm": 1.0546875,
"learning_rate": 0.00017300077281289845,
"loss": 1.5643,
"step": 775
},
{
"epoch": 0.8396124865446717,
"grad_norm": 0.8203125,
"learning_rate": 0.00017259392000922125,
"loss": 1.516,
"step": 780
},
{
"epoch": 0.844994617868676,
"grad_norm": 1.1328125,
"learning_rate": 0.0001721845107935556,
"loss": 1.6089,
"step": 785
},
{
"epoch": 0.8503767491926802,
"grad_norm": 0.67578125,
"learning_rate": 0.00017177255958334342,
"loss": 1.5203,
"step": 790
},
{
"epoch": 0.8557588805166846,
"grad_norm": 0.796875,
"learning_rate": 0.00017135808088554358,
"loss": 2.1079,
"step": 795
},
{
"epoch": 0.8611410118406889,
"grad_norm": 0.87890625,
"learning_rate": 0.000170941089296121,
"loss": 1.8979,
"step": 800
},
{
"epoch": 0.8665231431646933,
"grad_norm": 0.90625,
"learning_rate": 0.00017052159949953278,
"loss": 1.4225,
"step": 805
},
{
"epoch": 0.8719052744886975,
"grad_norm": 1.03125,
"learning_rate": 0.00017009962626821082,
"loss": 1.7616,
"step": 810
},
{
"epoch": 0.8772874058127018,
"grad_norm": 0.87890625,
"learning_rate": 0.0001696751844620419,
"loss": 1.8864,
"step": 815
},
{
"epoch": 0.8826695371367062,
"grad_norm": 1.234375,
"learning_rate": 0.00016924828902784407,
"loss": 1.6568,
"step": 820
},
{
"epoch": 0.8880516684607105,
"grad_norm": 0.8046875,
"learning_rate": 0.00016881895499884072,
"loss": 1.6826,
"step": 825
},
{
"epoch": 0.8934337997847147,
"grad_norm": 0.95703125,
"learning_rate": 0.00016838719749413063,
"loss": 1.3967,
"step": 830
},
{
"epoch": 0.898815931108719,
"grad_norm": 1.1953125,
"learning_rate": 0.00016795303171815616,
"loss": 1.7213,
"step": 835
},
{
"epoch": 0.9041980624327234,
"grad_norm": 0.8125,
"learning_rate": 0.00016751647296016725,
"loss": 1.6946,
"step": 840
},
{
"epoch": 0.9095801937567277,
"grad_norm": 0.71484375,
"learning_rate": 0.00016707753659368337,
"loss": 1.8869,
"step": 845
},
{
"epoch": 0.9149623250807319,
"grad_norm": 1.0,
"learning_rate": 0.0001666362380759521,
"loss": 1.5258,
"step": 850
},
{
"epoch": 0.9203444564047363,
"grad_norm": 1.28125,
"learning_rate": 0.0001661925929474046,
"loss": 1.8925,
"step": 855
},
{
"epoch": 0.9257265877287406,
"grad_norm": 0.84765625,
"learning_rate": 0.00016574661683110858,
"loss": 1.7461,
"step": 860
},
{
"epoch": 0.9311087190527448,
"grad_norm": 1.0546875,
"learning_rate": 0.00016529832543221796,
"loss": 1.6321,
"step": 865
},
{
"epoch": 0.9364908503767492,
"grad_norm": 0.83203125,
"learning_rate": 0.00016484773453741999,
"loss": 1.4394,
"step": 870
},
{
"epoch": 0.9418729817007535,
"grad_norm": 0.84375,
"learning_rate": 0.0001643948600143791,
"loss": 1.7612,
"step": 875
},
{
"epoch": 0.9472551130247578,
"grad_norm": 1.2890625,
"learning_rate": 0.00016393971781117827,
"loss": 1.5663,
"step": 880
},
{
"epoch": 0.9526372443487621,
"grad_norm": 0.9375,
"learning_rate": 0.00016348232395575738,
"loss": 1.6009,
"step": 885
},
{
"epoch": 0.9580193756727664,
"grad_norm": 0.796875,
"learning_rate": 0.0001630226945553487,
"loss": 1.5051,
"step": 890
},
{
"epoch": 0.9634015069967707,
"grad_norm": 0.58984375,
"learning_rate": 0.00016256084579590991,
"loss": 1.3798,
"step": 895
},
{
"epoch": 0.9687836383207751,
"grad_norm": 1.1328125,
"learning_rate": 0.00016209679394155378,
"loss": 1.542,
"step": 900
},
{
"epoch": 0.9741657696447793,
"grad_norm": 0.875,
"learning_rate": 0.0001616305553339756,
"loss": 1.7239,
"step": 905
},
{
"epoch": 0.9795479009687836,
"grad_norm": 1.0078125,
"learning_rate": 0.0001611621463918778,
"loss": 1.6649,
"step": 910
},
{
"epoch": 0.984930032292788,
"grad_norm": 0.91796875,
"learning_rate": 0.0001606915836103916,
"loss": 1.5315,
"step": 915
},
{
"epoch": 0.9903121636167922,
"grad_norm": 0.7734375,
"learning_rate": 0.00016021888356049607,
"loss": 1.6008,
"step": 920
},
{
"epoch": 0.9956942949407965,
"grad_norm": 0.875,
"learning_rate": 0.00015974406288843485,
"loss": 1.4304,
"step": 925
}
],
"logging_steps": 5,
"max_steps": 2787,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 1.6830365703340032e+16,
"train_batch_size": 16,
"trial_name": null,
"trial_params": null
}