| { |
| "best_global_step": 2500, |
| "best_metric": 0.3031768798828125, |
| "best_model_checkpoint": "./chemistry-qlora-optimized/checkpoint-2500", |
| "epoch": 0.29867240117676924, |
| "eval_steps": 500, |
| "global_step": 2500, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002986724011767693, |
| "grad_norm": 0.3601633906364441, |
| "learning_rate": 2.485089463220676e-05, |
| "loss": 0.9242, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.005973448023535386, |
| "grad_norm": 0.20105794072151184, |
| "learning_rate": 4.970178926441352e-05, |
| "loss": 0.5975, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.008960172035303077, |
| "grad_norm": 0.0951744094491005, |
| "learning_rate": 7.455268389662027e-05, |
| "loss": 0.4182, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.011946896047070771, |
| "grad_norm": 0.075951486825943, |
| "learning_rate": 9.940357852882704e-05, |
| "loss": 0.3922, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.014933620058838463, |
| "grad_norm": 0.07564502209424973, |
| "learning_rate": 0.0001242544731610338, |
| "loss": 0.3782, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.017920344070606154, |
| "grad_norm": 0.09620492905378342, |
| "learning_rate": 0.00014910536779324054, |
| "loss": 0.3748, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.020907068082373848, |
| "grad_norm": 0.07839681208133698, |
| "learning_rate": 0.00017395626242544733, |
| "loss": 0.3705, |
| "step": 175 |
| }, |
| { |
| "epoch": 0.023893792094141542, |
| "grad_norm": 0.08541898429393768, |
| "learning_rate": 0.00019880715705765408, |
| "loss": 0.3669, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.026880516105909233, |
| "grad_norm": 0.07105179876089096, |
| "learning_rate": 0.00022365805168986084, |
| "loss": 0.36, |
| "step": 225 |
| }, |
| { |
| "epoch": 0.029867240117676927, |
| "grad_norm": 0.07087098807096481, |
| "learning_rate": 0.0002485089463220676, |
| "loss": 0.3606, |
| "step": 250 |
| }, |
| { |
| "epoch": 0.03285396412944462, |
| "grad_norm": 0.07232872396707535, |
| "learning_rate": 0.00027335984095427435, |
| "loss": 0.3572, |
| "step": 275 |
| }, |
| { |
| "epoch": 0.03584068814121231, |
| "grad_norm": 0.0895916074514389, |
| "learning_rate": 0.0002982107355864811, |
| "loss": 0.3537, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.03882741215298, |
| "grad_norm": 0.06402889639139175, |
| "learning_rate": 0.0003230616302186879, |
| "loss": 0.3618, |
| "step": 325 |
| }, |
| { |
| "epoch": 0.041814136164747696, |
| "grad_norm": 0.055026255548000336, |
| "learning_rate": 0.00034791252485089465, |
| "loss": 0.3504, |
| "step": 350 |
| }, |
| { |
| "epoch": 0.04480086017651539, |
| "grad_norm": 0.06303983181715012, |
| "learning_rate": 0.0003727634194831014, |
| "loss": 0.3526, |
| "step": 375 |
| }, |
| { |
| "epoch": 0.047787584188283085, |
| "grad_norm": 0.06284493952989578, |
| "learning_rate": 0.00039761431411530816, |
| "loss": 0.3467, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.05077430820005077, |
| "grad_norm": 0.05819180980324745, |
| "learning_rate": 0.00042246520874751495, |
| "loss": 0.3506, |
| "step": 425 |
| }, |
| { |
| "epoch": 0.053761032211818466, |
| "grad_norm": 0.08199992030858994, |
| "learning_rate": 0.0004473161033797217, |
| "loss": 0.3453, |
| "step": 450 |
| }, |
| { |
| "epoch": 0.05674775622358616, |
| "grad_norm": 0.06815178692340851, |
| "learning_rate": 0.0004721669980119284, |
| "loss": 0.3427, |
| "step": 475 |
| }, |
| { |
| "epoch": 0.059734480235353854, |
| "grad_norm": 0.050426531583070755, |
| "learning_rate": 0.0004970178926441352, |
| "loss": 0.3441, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.059734480235353854, |
| "eval_loss": 0.3422516882419586, |
| "eval_runtime": 80020.5739, |
| "eval_samples_per_second": 3.347, |
| "eval_steps_per_second": 0.418, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.06272120424712155, |
| "grad_norm": 0.08422357589006424, |
| "learning_rate": 0.0004993225349510377, |
| "loss": 0.3449, |
| "step": 525 |
| }, |
| { |
| "epoch": 0.06570792825888924, |
| "grad_norm": 0.09289127588272095, |
| "learning_rate": 0.0004985526883044898, |
| "loss": 0.338, |
| "step": 550 |
| }, |
| { |
| "epoch": 0.06869465227065694, |
| "grad_norm": 0.13212044537067413, |
| "learning_rate": 0.0004977828416579417, |
| "loss": 0.3402, |
| "step": 575 |
| }, |
| { |
| "epoch": 0.07168137628242462, |
| "grad_norm": 0.0823817029595375, |
| "learning_rate": 0.0004970129950113937, |
| "loss": 0.3372, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.07466810029419231, |
| "grad_norm": 0.05125906690955162, |
| "learning_rate": 0.0004962431483648457, |
| "loss": 0.336, |
| "step": 625 |
| }, |
| { |
| "epoch": 0.07765482430596, |
| "grad_norm": 0.267816424369812, |
| "learning_rate": 0.0004954733017182978, |
| "loss": 0.34, |
| "step": 650 |
| }, |
| { |
| "epoch": 0.0806415483177277, |
| "grad_norm": 0.057936109602451324, |
| "learning_rate": 0.0004947034550717497, |
| "loss": 0.3364, |
| "step": 675 |
| }, |
| { |
| "epoch": 0.08362827232949539, |
| "grad_norm": 0.050852783024311066, |
| "learning_rate": 0.0004939336084252017, |
| "loss": 0.3353, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.08661499634126309, |
| "grad_norm": 0.04943128302693367, |
| "learning_rate": 0.0004931637617786536, |
| "loss": 0.3284, |
| "step": 725 |
| }, |
| { |
| "epoch": 0.08960172035303078, |
| "grad_norm": 0.04955856129527092, |
| "learning_rate": 0.0004923939151321057, |
| "loss": 0.3285, |
| "step": 750 |
| }, |
| { |
| "epoch": 0.09258844436479848, |
| "grad_norm": 0.04910452291369438, |
| "learning_rate": 0.0004916240684855577, |
| "loss": 0.3291, |
| "step": 775 |
| }, |
| { |
| "epoch": 0.09557516837656617, |
| "grad_norm": 0.05659394711256027, |
| "learning_rate": 0.0004908542218390097, |
| "loss": 0.3283, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.09856189238833385, |
| "grad_norm": 0.05103044584393501, |
| "learning_rate": 0.0004900843751924616, |
| "loss": 0.329, |
| "step": 825 |
| }, |
| { |
| "epoch": 0.10154861640010154, |
| "grad_norm": 0.07405893504619598, |
| "learning_rate": 0.0004893145285459137, |
| "loss": 0.3266, |
| "step": 850 |
| }, |
| { |
| "epoch": 0.10453534041186924, |
| "grad_norm": 0.04762987419962883, |
| "learning_rate": 0.0004885446818993657, |
| "loss": 0.3249, |
| "step": 875 |
| }, |
| { |
| "epoch": 0.10752206442363693, |
| "grad_norm": 0.050489380955696106, |
| "learning_rate": 0.0004877748352528176, |
| "loss": 0.3244, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.11050878843540463, |
| "grad_norm": 0.049161575734615326, |
| "learning_rate": 0.00048700498860626963, |
| "loss": 0.3275, |
| "step": 925 |
| }, |
| { |
| "epoch": 0.11349551244717232, |
| "grad_norm": 0.0867818295955658, |
| "learning_rate": 0.0004862351419597216, |
| "loss": 0.3207, |
| "step": 950 |
| }, |
| { |
| "epoch": 0.11648223645894001, |
| "grad_norm": 0.049138959497213364, |
| "learning_rate": 0.00048546529531317363, |
| "loss": 0.3203, |
| "step": 975 |
| }, |
| { |
| "epoch": 0.11946896047070771, |
| "grad_norm": 0.049613844603300095, |
| "learning_rate": 0.0004846954486666256, |
| "loss": 0.321, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.11946896047070771, |
| "eval_loss": 0.32202616333961487, |
| "eval_runtime": 80142.6561, |
| "eval_samples_per_second": 3.342, |
| "eval_steps_per_second": 0.418, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.1224556844824754, |
| "grad_norm": 0.05376691371202469, |
| "learning_rate": 0.00048392560202007763, |
| "loss": 0.3184, |
| "step": 1025 |
| }, |
| { |
| "epoch": 0.1254424084942431, |
| "grad_norm": 0.048957519233226776, |
| "learning_rate": 0.0004831557553735296, |
| "loss": 0.3193, |
| "step": 1050 |
| }, |
| { |
| "epoch": 0.1284291325060108, |
| "grad_norm": 0.0559055432677269, |
| "learning_rate": 0.00048238590872698157, |
| "loss": 0.3261, |
| "step": 1075 |
| }, |
| { |
| "epoch": 0.13141585651777848, |
| "grad_norm": 0.06237791106104851, |
| "learning_rate": 0.00048161606208043354, |
| "loss": 0.3235, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.13440258052954618, |
| "grad_norm": 0.05047965049743652, |
| "learning_rate": 0.00048084621543388557, |
| "loss": 0.3232, |
| "step": 1125 |
| }, |
| { |
| "epoch": 0.13738930454131387, |
| "grad_norm": 0.05053680017590523, |
| "learning_rate": 0.00048007636878733754, |
| "loss": 0.3166, |
| "step": 1150 |
| }, |
| { |
| "epoch": 0.14037602855308154, |
| "grad_norm": 0.1490963250398636, |
| "learning_rate": 0.00047930652214078957, |
| "loss": 0.3196, |
| "step": 1175 |
| }, |
| { |
| "epoch": 0.14336275256484923, |
| "grad_norm": 0.06572224199771881, |
| "learning_rate": 0.00047853667549424154, |
| "loss": 0.3162, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.14634947657661693, |
| "grad_norm": 0.04761017858982086, |
| "learning_rate": 0.00047776682884769357, |
| "loss": 0.3188, |
| "step": 1225 |
| }, |
| { |
| "epoch": 0.14933620058838462, |
| "grad_norm": 0.05511182174086571, |
| "learning_rate": 0.00047699698220114554, |
| "loss": 0.3211, |
| "step": 1250 |
| }, |
| { |
| "epoch": 0.15232292460015232, |
| "grad_norm": 0.05186431482434273, |
| "learning_rate": 0.0004762271355545975, |
| "loss": 0.3166, |
| "step": 1275 |
| }, |
| { |
| "epoch": 0.15530964861192, |
| "grad_norm": 0.05169636383652687, |
| "learning_rate": 0.0004754572889080495, |
| "loss": 0.3161, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.1582963726236877, |
| "grad_norm": 0.05161893740296364, |
| "learning_rate": 0.0004746874422615015, |
| "loss": 0.3106, |
| "step": 1325 |
| }, |
| { |
| "epoch": 0.1612830966354554, |
| "grad_norm": 0.047998517751693726, |
| "learning_rate": 0.0004739175956149535, |
| "loss": 0.3092, |
| "step": 1350 |
| }, |
| { |
| "epoch": 0.1642698206472231, |
| "grad_norm": 0.05086188018321991, |
| "learning_rate": 0.0004731477489684055, |
| "loss": 0.314, |
| "step": 1375 |
| }, |
| { |
| "epoch": 0.16725654465899079, |
| "grad_norm": 0.5123017430305481, |
| "learning_rate": 0.0004723779023218575, |
| "loss": 0.3138, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.17024326867075848, |
| "grad_norm": 0.513543426990509, |
| "learning_rate": 0.0004716080556753095, |
| "loss": 0.3477, |
| "step": 1425 |
| }, |
| { |
| "epoch": 0.17322999268252617, |
| "grad_norm": 0.06812719255685806, |
| "learning_rate": 0.00047083820902876143, |
| "loss": 0.3183, |
| "step": 1450 |
| }, |
| { |
| "epoch": 0.17621671669429387, |
| "grad_norm": 0.053058214485645294, |
| "learning_rate": 0.00047006836238221346, |
| "loss": 0.3132, |
| "step": 1475 |
| }, |
| { |
| "epoch": 0.17920344070606156, |
| "grad_norm": 0.05340981110930443, |
| "learning_rate": 0.00046929851573566543, |
| "loss": 0.3167, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.17920344070606156, |
| "eval_loss": 0.31384027004241943, |
| "eval_runtime": 79940.0478, |
| "eval_samples_per_second": 3.351, |
| "eval_steps_per_second": 0.419, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.18219016471782926, |
| "grad_norm": 0.06443135440349579, |
| "learning_rate": 0.00046852866908911746, |
| "loss": 0.3088, |
| "step": 1525 |
| }, |
| { |
| "epoch": 0.18517688872959695, |
| "grad_norm": 0.05341175198554993, |
| "learning_rate": 0.00046775882244256943, |
| "loss": 0.3131, |
| "step": 1550 |
| }, |
| { |
| "epoch": 0.18816361274136464, |
| "grad_norm": 0.05056153982877731, |
| "learning_rate": 0.00046698897579602145, |
| "loss": 0.3101, |
| "step": 1575 |
| }, |
| { |
| "epoch": 0.19115033675313234, |
| "grad_norm": 0.05725909024477005, |
| "learning_rate": 0.0004662191291494734, |
| "loss": 0.3086, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.19413706076490003, |
| "grad_norm": 0.473294198513031, |
| "learning_rate": 0.0004654800763687874, |
| "loss": 0.3339, |
| "step": 1625 |
| }, |
| { |
| "epoch": 0.1971237847766677, |
| "grad_norm": 0.5311633944511414, |
| "learning_rate": 0.00046471022972223935, |
| "loss": 0.3326, |
| "step": 1650 |
| }, |
| { |
| "epoch": 0.2001105087884354, |
| "grad_norm": 0.05935530364513397, |
| "learning_rate": 0.0004639403830756913, |
| "loss": 0.3207, |
| "step": 1675 |
| }, |
| { |
| "epoch": 0.2030972328002031, |
| "grad_norm": 0.0760846734046936, |
| "learning_rate": 0.0004631705364291433, |
| "loss": 0.3103, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.20608395681197078, |
| "grad_norm": 0.05102820321917534, |
| "learning_rate": 0.0004624006897825953, |
| "loss": 0.3103, |
| "step": 1725 |
| }, |
| { |
| "epoch": 0.20907068082373848, |
| "grad_norm": 0.05322102829813957, |
| "learning_rate": 0.0004616308431360473, |
| "loss": 0.3107, |
| "step": 1750 |
| }, |
| { |
| "epoch": 0.21205740483550617, |
| "grad_norm": 0.05677218735218048, |
| "learning_rate": 0.0004608609964894993, |
| "loss": 0.3107, |
| "step": 1775 |
| }, |
| { |
| "epoch": 0.21504412884727386, |
| "grad_norm": 0.05003349855542183, |
| "learning_rate": 0.0004600911498429513, |
| "loss": 0.307, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.21803085285904156, |
| "grad_norm": 0.05029498040676117, |
| "learning_rate": 0.0004593213031964033, |
| "loss": 0.3054, |
| "step": 1825 |
| }, |
| { |
| "epoch": 0.22101757687080925, |
| "grad_norm": 0.05433105677366257, |
| "learning_rate": 0.0004585514565498553, |
| "loss": 0.3108, |
| "step": 1850 |
| }, |
| { |
| "epoch": 0.22400430088257695, |
| "grad_norm": 0.06716553121805191, |
| "learning_rate": 0.00045778160990330726, |
| "loss": 0.3089, |
| "step": 1875 |
| }, |
| { |
| "epoch": 0.22699102489434464, |
| "grad_norm": 0.06449099630117416, |
| "learning_rate": 0.00045701176325675923, |
| "loss": 0.3081, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.22997774890611233, |
| "grad_norm": 0.060889799147844315, |
| "learning_rate": 0.00045624191661021126, |
| "loss": 0.3085, |
| "step": 1925 |
| }, |
| { |
| "epoch": 0.23296447291788003, |
| "grad_norm": 0.13389873504638672, |
| "learning_rate": 0.00045547206996366323, |
| "loss": 0.3048, |
| "step": 1950 |
| }, |
| { |
| "epoch": 0.23595119692964772, |
| "grad_norm": 0.05601953715085983, |
| "learning_rate": 0.00045470222331711526, |
| "loss": 0.3063, |
| "step": 1975 |
| }, |
| { |
| "epoch": 0.23893792094141542, |
| "grad_norm": 0.05448105186223984, |
| "learning_rate": 0.00045393237667056723, |
| "loss": 0.3033, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.23893792094141542, |
| "eval_loss": 0.3051091432571411, |
| "eval_runtime": 80198.5724, |
| "eval_samples_per_second": 3.34, |
| "eval_steps_per_second": 0.417, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.2419246449531831, |
| "grad_norm": 0.0593348927795887, |
| "learning_rate": 0.00045316253002401926, |
| "loss": 0.3033, |
| "step": 2025 |
| }, |
| { |
| "epoch": 0.2449113689649508, |
| "grad_norm": 0.0662638247013092, |
| "learning_rate": 0.0004523926833774712, |
| "loss": 0.3028, |
| "step": 2050 |
| }, |
| { |
| "epoch": 0.2478980929767185, |
| "grad_norm": 0.08647308498620987, |
| "learning_rate": 0.0004516228367309232, |
| "loss": 0.3015, |
| "step": 2075 |
| }, |
| { |
| "epoch": 0.2508848169884862, |
| "grad_norm": 0.06536151468753815, |
| "learning_rate": 0.0004508529900843752, |
| "loss": 0.3031, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.2538715410002539, |
| "grad_norm": 0.05942423641681671, |
| "learning_rate": 0.0004500831434378272, |
| "loss": 0.3037, |
| "step": 2125 |
| }, |
| { |
| "epoch": 0.2568582650120216, |
| "grad_norm": 0.05728958174586296, |
| "learning_rate": 0.0004493132967912792, |
| "loss": 0.3041, |
| "step": 2150 |
| }, |
| { |
| "epoch": 0.2598449890237893, |
| "grad_norm": 0.07348435372114182, |
| "learning_rate": 0.0004485434501447312, |
| "loss": 0.3054, |
| "step": 2175 |
| }, |
| { |
| "epoch": 0.26283171303555697, |
| "grad_norm": 0.07285642623901367, |
| "learning_rate": 0.0004477736034981832, |
| "loss": 0.3012, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.26581843704732466, |
| "grad_norm": 0.06905517727136612, |
| "learning_rate": 0.0004470037568516352, |
| "loss": 0.3002, |
| "step": 2225 |
| }, |
| { |
| "epoch": 0.26880516105909236, |
| "grad_norm": 0.047947004437446594, |
| "learning_rate": 0.0004462339102050871, |
| "loss": 0.3013, |
| "step": 2250 |
| }, |
| { |
| "epoch": 0.27179188507086005, |
| "grad_norm": 0.12290674448013306, |
| "learning_rate": 0.00044546406355853914, |
| "loss": 0.3012, |
| "step": 2275 |
| }, |
| { |
| "epoch": 0.27477860908262774, |
| "grad_norm": 0.05509233847260475, |
| "learning_rate": 0.0004446942169119911, |
| "loss": 0.2998, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.27776533309439544, |
| "grad_norm": 0.06934493780136108, |
| "learning_rate": 0.00044392437026544314, |
| "loss": 0.3014, |
| "step": 2325 |
| }, |
| { |
| "epoch": 0.2807520571061631, |
| "grad_norm": 0.1120632141828537, |
| "learning_rate": 0.0004431545236188951, |
| "loss": 0.299, |
| "step": 2350 |
| }, |
| { |
| "epoch": 0.28373878111793077, |
| "grad_norm": 0.0867624506354332, |
| "learning_rate": 0.00044238467697234714, |
| "loss": 0.2983, |
| "step": 2375 |
| }, |
| { |
| "epoch": 0.28672550512969847, |
| "grad_norm": 0.07918841391801834, |
| "learning_rate": 0.0004416148303257991, |
| "loss": 0.3006, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.28971222914146616, |
| "grad_norm": 0.08205003291368484, |
| "learning_rate": 0.0004408449836792511, |
| "loss": 0.3008, |
| "step": 2425 |
| }, |
| { |
| "epoch": 0.29269895315323385, |
| "grad_norm": 0.05923304334282875, |
| "learning_rate": 0.00044007513703270306, |
| "loss": 0.2984, |
| "step": 2450 |
| }, |
| { |
| "epoch": 0.29568567716500155, |
| "grad_norm": 0.06517321616411209, |
| "learning_rate": 0.0004393052903861551, |
| "loss": 0.3024, |
| "step": 2475 |
| }, |
| { |
| "epoch": 0.29867240117676924, |
| "grad_norm": 0.09344199299812317, |
| "learning_rate": 0.00043853544373960706, |
| "loss": 0.2991, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.29867240117676924, |
| "eval_loss": 0.3031768798828125, |
| "eval_runtime": 80460.0276, |
| "eval_samples_per_second": 3.329, |
| "eval_steps_per_second": 0.416, |
| "step": 2500 |
| } |
| ], |
| "logging_steps": 25, |
| "max_steps": 16740, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 2, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "EarlyStoppingCallback": { |
| "args": { |
| "early_stopping_patience": 3, |
| "early_stopping_threshold": 0.01 |
| }, |
| "attributes": { |
| "early_stopping_patience_counter": 3 |
| } |
| }, |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1.306916403806208e+19, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|