| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.8, |
| "eval_steps": 1000, |
| "global_step": 12000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.006666666666666667, |
| "grad_norm": 2.854929208755493, |
| "learning_rate": 5.28e-05, |
| "loss": 18.5724072265625, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.013333333333333334, |
| "grad_norm": 1.6583245992660522, |
| "learning_rate": 0.00010613333333333333, |
| "loss": 14.702509765625, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 1.6954909563064575, |
| "learning_rate": 0.00015946666666666668, |
| "loss": 13.69074951171875, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.02666666666666667, |
| "grad_norm": 1.783247470855713, |
| "learning_rate": 0.00021280000000000002, |
| "loss": 13.1521044921875, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.03333333333333333, |
| "grad_norm": 1.4421945810317993, |
| "learning_rate": 0.00026613333333333337, |
| "loss": 12.824429931640625, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 1.5354344844818115, |
| "learning_rate": 0.00031946666666666666, |
| "loss": 12.49961181640625, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.04666666666666667, |
| "grad_norm": 1.3272199630737305, |
| "learning_rate": 0.00037280000000000006, |
| "loss": 12.145262451171876, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.05333333333333334, |
| "grad_norm": 1.706987977027893, |
| "learning_rate": 0.0003999883303467061, |
| "loss": 11.98048095703125, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 1.2479976415634155, |
| "learning_rate": 0.00039989210445800615, |
| "loss": 11.621953125, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.06666666666666667, |
| "grad_norm": 1.2310028076171875, |
| "learning_rate": 0.00039969872739228826, |
| "loss": 11.573458251953125, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.06666666666666667, |
| "eval_accuracy": 0.18273972602739727, |
| "eval_loss": 11.462567329406738, |
| "eval_runtime": 65.4243, |
| "eval_samples_per_second": 7.642, |
| "eval_steps_per_second": 0.382, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.07333333333333333, |
| "grad_norm": 1.2715719938278198, |
| "learning_rate": 0.00039940829313430293, |
| "loss": 11.32210205078125, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 1.7137954235076904, |
| "learning_rate": 0.00039902094284035075, |
| "loss": 11.18190185546875, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.08666666666666667, |
| "grad_norm": 1.48090660572052, |
| "learning_rate": 0.0003985368647696784, |
| "loss": 11.0220166015625, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.09333333333333334, |
| "grad_norm": 1.1374964714050293, |
| "learning_rate": 0.0003979562941929808, |
| "loss": 10.822305908203125, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 1.3714447021484375, |
| "learning_rate": 0.0003972795132780554, |
| "loss": 10.762421875, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.10666666666666667, |
| "grad_norm": 1.1782511472702026, |
| "learning_rate": 0.00039650685095266405, |
| "loss": 10.602275390625, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.11333333333333333, |
| "grad_norm": 1.0709394216537476, |
| "learning_rate": 0.0003956386827446671, |
| "loss": 10.4713134765625, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 1.1138310432434082, |
| "learning_rate": 0.00039467543059951106, |
| "loss": 10.50548095703125, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.12666666666666668, |
| "grad_norm": 1.283762812614441, |
| "learning_rate": 0.0003936175626751552, |
| "loss": 10.4138134765625, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.13333333333333333, |
| "grad_norm": 1.3275446891784668, |
| "learning_rate": 0.00039246559311453794, |
| "loss": 10.335478515625, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.13333333333333333, |
| "eval_accuracy": 0.21337964774951076, |
| "eval_loss": 10.411324501037598, |
| "eval_runtime": 65.3972, |
| "eval_samples_per_second": 7.646, |
| "eval_steps_per_second": 0.382, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 1.456148386001587, |
| "learning_rate": 0.00039122008179569447, |
| "loss": 10.4031689453125, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.14666666666666667, |
| "grad_norm": 1.4301263093948364, |
| "learning_rate": 0.00038988163405964595, |
| "loss": 10.173363037109375, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.15333333333333332, |
| "grad_norm": 1.0371617078781128, |
| "learning_rate": 0.00038845090041619265, |
| "loss": 10.063040771484374, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 1.0449143648147583, |
| "learning_rate": 0.0003869285762277543, |
| "loss": 10.145305786132813, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.16666666666666666, |
| "grad_norm": 1.491047739982605, |
| "learning_rate": 0.00038531540137141165, |
| "loss": 10.095411376953125, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.17333333333333334, |
| "grad_norm": 1.216232419013977, |
| "learning_rate": 0.0003836121598793126, |
| "loss": 10.10360107421875, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.18, |
| "grad_norm": 1.0485482215881348, |
| "learning_rate": 0.0003818196795576189, |
| "loss": 9.9664892578125, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.18666666666666668, |
| "grad_norm": 1.1029548645019531, |
| "learning_rate": 0.00037993883158417654, |
| "loss": 9.9088525390625, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.19333333333333333, |
| "grad_norm": 1.0070501565933228, |
| "learning_rate": 0.00037797053008510834, |
| "loss": 9.888450927734375, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 1.1356112957000732, |
| "learning_rate": 0.00037591573169053167, |
| "loss": 9.8390869140625, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.2, |
| "eval_accuracy": 0.2266908023483366, |
| "eval_loss": 9.943662643432617, |
| "eval_runtime": 65.3824, |
| "eval_samples_per_second": 7.647, |
| "eval_steps_per_second": 0.382, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.20666666666666667, |
| "grad_norm": 1.178817629814148, |
| "learning_rate": 0.0003737754350696192, |
| "loss": 9.806249389648437, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.21333333333333335, |
| "grad_norm": 1.1421138048171997, |
| "learning_rate": 0.00037155068044522735, |
| "loss": 9.715457763671875, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.22, |
| "grad_norm": 1.6934456825256348, |
| "learning_rate": 0.00036924254908832937, |
| "loss": 9.71255859375, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.22666666666666666, |
| "grad_norm": 1.1598275899887085, |
| "learning_rate": 0.00036685216279249815, |
| "loss": 9.716998291015624, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.23333333333333334, |
| "grad_norm": 1.1768417358398438, |
| "learning_rate": 0.000364380683328694, |
| "loss": 9.557822875976562, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.24, |
| "grad_norm": 1.4840753078460693, |
| "learning_rate": 0.0003618293118806232, |
| "loss": 9.56420654296875, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.24666666666666667, |
| "grad_norm": 1.2799503803253174, |
| "learning_rate": 0.00035919928846094094, |
| "loss": 9.548779296875, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.25333333333333335, |
| "grad_norm": 1.1078740358352661, |
| "learning_rate": 0.00035649189130858256, |
| "loss": 9.44154541015625, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.26, |
| "grad_norm": 1.173609972000122, |
| "learning_rate": 0.00035370843626751663, |
| "loss": 9.414360961914063, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.26666666666666666, |
| "grad_norm": 1.2394672632217407, |
| "learning_rate": 0.00035085027614722075, |
| "loss": 9.300681762695312, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.26666666666666666, |
| "eval_accuracy": 0.2545146771037182, |
| "eval_loss": 9.379138946533203, |
| "eval_runtime": 65.4591, |
| "eval_samples_per_second": 7.638, |
| "eval_steps_per_second": 0.382, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.2733333333333333, |
| "grad_norm": 1.0034714937210083, |
| "learning_rate": 0.00034791880006519195, |
| "loss": 9.250709838867188, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.28, |
| "grad_norm": 1.2436392307281494, |
| "learning_rate": 0.00034491543277181, |
| "loss": 9.246346435546876, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.2866666666666667, |
| "grad_norm": 1.180080771446228, |
| "learning_rate": 0.00034184163395788343, |
| "loss": 9.188368530273438, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.29333333333333333, |
| "grad_norm": 1.0143563747406006, |
| "learning_rate": 0.00033869889754521314, |
| "loss": 9.10044677734375, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.3, |
| "grad_norm": 1.2389507293701172, |
| "learning_rate": 0.0003354887509605197, |
| "loss": 9.080850830078125, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.30666666666666664, |
| "grad_norm": 0.997056782245636, |
| "learning_rate": 0.0003322127543930859, |
| "loss": 9.041187744140625, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.31333333333333335, |
| "grad_norm": 1.006325364112854, |
| "learning_rate": 0.00032887250003647676, |
| "loss": 9.152498779296875, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.32, |
| "grad_norm": 1.1522862911224365, |
| "learning_rate": 0.00032546961131470485, |
| "loss": 8.979527587890624, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.32666666666666666, |
| "grad_norm": 0.9992244839668274, |
| "learning_rate": 0.00032200574209321657, |
| "loss": 8.930678100585938, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.3333333333333333, |
| "grad_norm": 1.188022255897522, |
| "learning_rate": 0.0003184825758750839, |
| "loss": 8.919720458984376, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.3333333333333333, |
| "eval_accuracy": 0.27488649706457924, |
| "eval_loss": 8.979753494262695, |
| "eval_runtime": 65.4227, |
| "eval_samples_per_second": 7.643, |
| "eval_steps_per_second": 0.382, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.34, |
| "grad_norm": 0.9783419966697693, |
| "learning_rate": 0.00031490182498279115, |
| "loss": 8.924981079101563, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.3466666666666667, |
| "grad_norm": 1.0431445837020874, |
| "learning_rate": 0.0003112652297260157, |
| "loss": 8.822501220703124, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.35333333333333333, |
| "grad_norm": 0.9755280613899231, |
| "learning_rate": 0.00030757455755580553, |
| "loss": 8.865919189453125, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.36, |
| "grad_norm": 1.1168733835220337, |
| "learning_rate": 0.0003038316022055665, |
| "loss": 8.899959716796875, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.36666666666666664, |
| "grad_norm": 0.9611853957176208, |
| "learning_rate": 0.00030003818281927526, |
| "loss": 8.851053466796875, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.37333333333333335, |
| "grad_norm": 1.040235996246338, |
| "learning_rate": 0.00029619614306734235, |
| "loss": 8.790068359375, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.38, |
| "grad_norm": 1.0123136043548584, |
| "learning_rate": 0.00029230735025055524, |
| "loss": 8.8330029296875, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.38666666666666666, |
| "grad_norm": 1.0311475992202759, |
| "learning_rate": 0.00028837369439253617, |
| "loss": 8.824832153320312, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.3933333333333333, |
| "grad_norm": 0.997328519821167, |
| "learning_rate": 0.0002843970873211566, |
| "loss": 8.828201904296876, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 1.5238865613937378, |
| "learning_rate": 0.0002803794617393543, |
| "loss": 8.719149169921875, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.4, |
| "eval_accuracy": 0.28737377690802346, |
| "eval_loss": 8.731175422668457, |
| "eval_runtime": 65.6219, |
| "eval_samples_per_second": 7.619, |
| "eval_steps_per_second": 0.381, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.4066666666666667, |
| "grad_norm": 1.1989500522613525, |
| "learning_rate": 0.0002763227702858047, |
| "loss": 8.64713623046875, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.41333333333333333, |
| "grad_norm": 1.3336397409439087, |
| "learning_rate": 0.00027222898458590343, |
| "loss": 8.749034423828125, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.42, |
| "grad_norm": 1.0257115364074707, |
| "learning_rate": 0.0002681000942935204, |
| "loss": 8.847327880859375, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.4266666666666667, |
| "grad_norm": 1.03694748878479, |
| "learning_rate": 0.0002639381061239921, |
| "loss": 8.704371337890626, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.43333333333333335, |
| "grad_norm": 0.9518272876739502, |
| "learning_rate": 0.00025974504287882194, |
| "loss": 8.664287109375, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.44, |
| "grad_norm": 1.0655875205993652, |
| "learning_rate": 0.000255522942462562, |
| "loss": 8.662864990234375, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.44666666666666666, |
| "grad_norm": 1.0382905006408691, |
| "learning_rate": 0.00025127385689235426, |
| "loss": 8.6200341796875, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.4533333333333333, |
| "grad_norm": 1.3670793771743774, |
| "learning_rate": 0.00024699985130061374, |
| "loss": 8.617677612304687, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.46, |
| "grad_norm": 1.2171597480773926, |
| "learning_rate": 0.0002427030029313362, |
| "loss": 8.605474853515625, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.4666666666666667, |
| "grad_norm": 1.1724241971969604, |
| "learning_rate": 0.00023838540013052062, |
| "loss": 8.558206787109375, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.4666666666666667, |
| "eval_accuracy": 0.2943894324853229, |
| "eval_loss": 8.570584297180176, |
| "eval_runtime": 65.4101, |
| "eval_samples_per_second": 7.644, |
| "eval_steps_per_second": 0.382, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.47333333333333333, |
| "grad_norm": 1.0969085693359375, |
| "learning_rate": 0.00023404914133119486, |
| "loss": 8.515238037109375, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.48, |
| "grad_norm": 1.4420628547668457, |
| "learning_rate": 0.00022969633403353913, |
| "loss": 8.62325927734375, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.4866666666666667, |
| "grad_norm": 1.0490083694458008, |
| "learning_rate": 0.0002253290937806034, |
| "loss": 8.469694213867188, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.49333333333333335, |
| "grad_norm": 1.3232401609420776, |
| "learning_rate": 0.00022094954313011468, |
| "loss": 8.4819580078125, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.5, |
| "grad_norm": 1.0941740274429321, |
| "learning_rate": 0.0002165598106228758, |
| "loss": 8.58266357421875, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.5066666666666667, |
| "grad_norm": 1.1495511531829834, |
| "learning_rate": 0.00021216202974825614, |
| "loss": 8.533701171875, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.5133333333333333, |
| "grad_norm": 1.004111886024475, |
| "learning_rate": 0.00020775833790727695, |
| "loss": 8.46027099609375, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.52, |
| "grad_norm": 1.4242042303085327, |
| "learning_rate": 0.0002033508753737963, |
| "loss": 8.484051513671876, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.5266666666666666, |
| "grad_norm": 1.1219866275787354, |
| "learning_rate": 0.00019894178425429674, |
| "loss": 8.434028930664063, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.5333333333333333, |
| "grad_norm": 1.009364128112793, |
| "learning_rate": 0.00019453320744678324, |
| "loss": 8.4456640625, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.5333333333333333, |
| "eval_accuracy": 0.29964774951076323, |
| "eval_loss": 8.433549880981445, |
| "eval_runtime": 65.652, |
| "eval_samples_per_second": 7.616, |
| "eval_steps_per_second": 0.381, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.54, |
| "grad_norm": 0.9676337838172913, |
| "learning_rate": 0.0001901272875992957, |
| "loss": 8.395182495117188, |
| "step": 8100 |
| }, |
| { |
| "epoch": 0.5466666666666666, |
| "grad_norm": 1.0351442098617554, |
| "learning_rate": 0.0001857261660685435, |
| "loss": 8.413995361328125, |
| "step": 8200 |
| }, |
| { |
| "epoch": 0.5533333333333333, |
| "grad_norm": 0.987281322479248, |
| "learning_rate": 0.00018133198187916734, |
| "loss": 8.399580688476563, |
| "step": 8300 |
| }, |
| { |
| "epoch": 0.56, |
| "grad_norm": 0.9454151391983032, |
| "learning_rate": 0.00017694687068413458, |
| "loss": 8.42725341796875, |
| "step": 8400 |
| }, |
| { |
| "epoch": 0.5666666666666667, |
| "grad_norm": 1.0215506553649902, |
| "learning_rate": 0.00017257296372677328, |
| "loss": 8.389710693359374, |
| "step": 8500 |
| }, |
| { |
| "epoch": 0.5733333333333334, |
| "grad_norm": 1.0646560192108154, |
| "learning_rate": 0.00016821238680494918, |
| "loss": 8.27107177734375, |
| "step": 8600 |
| }, |
| { |
| "epoch": 0.58, |
| "grad_norm": 1.016783356666565, |
| "learning_rate": 0.00016386725923789007, |
| "loss": 8.3458056640625, |
| "step": 8700 |
| }, |
| { |
| "epoch": 0.5866666666666667, |
| "grad_norm": 1.029847502708435, |
| "learning_rate": 0.00015953969283615782, |
| "loss": 8.26437744140625, |
| "step": 8800 |
| }, |
| { |
| "epoch": 0.5933333333333334, |
| "grad_norm": 0.982424259185791, |
| "learning_rate": 0.0001552317908752705, |
| "loss": 8.396544189453126, |
| "step": 8900 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 0.958069384098053, |
| "learning_rate": 0.0001509456470734723, |
| "loss": 8.241473388671874, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.6, |
| "eval_accuracy": 0.3059334637964775, |
| "eval_loss": 8.307632446289062, |
| "eval_runtime": 65.4278, |
| "eval_samples_per_second": 7.642, |
| "eval_steps_per_second": 0.382, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.6066666666666667, |
| "grad_norm": 1.1917734146118164, |
| "learning_rate": 0.0001466833445741486, |
| "loss": 8.325388793945313, |
| "step": 9100 |
| }, |
| { |
| "epoch": 0.6133333333333333, |
| "grad_norm": 1.0918893814086914, |
| "learning_rate": 0.0001424469549333809, |
| "loss": 8.24582763671875, |
| "step": 9200 |
| }, |
| { |
| "epoch": 0.62, |
| "grad_norm": 0.9687617421150208, |
| "learning_rate": 0.0001382385371131328, |
| "loss": 8.22963134765625, |
| "step": 9300 |
| }, |
| { |
| "epoch": 0.6266666666666667, |
| "grad_norm": 1.2594196796417236, |
| "learning_rate": 0.0001340601364805574, |
| "loss": 8.203480224609375, |
| "step": 9400 |
| }, |
| { |
| "epoch": 0.6333333333333333, |
| "grad_norm": 1.327571153640747, |
| "learning_rate": 0.00012991378381391192, |
| "loss": 8.20176513671875, |
| "step": 9500 |
| }, |
| { |
| "epoch": 0.64, |
| "grad_norm": 1.0071412324905396, |
| "learning_rate": 0.0001258014943155624, |
| "loss": 8.172748413085937, |
| "step": 9600 |
| }, |
| { |
| "epoch": 0.6466666666666666, |
| "grad_norm": 1.0164071321487427, |
| "learning_rate": 0.00012172526663255951, |
| "loss": 8.214456787109375, |
| "step": 9700 |
| }, |
| { |
| "epoch": 0.6533333333333333, |
| "grad_norm": 1.1246143579483032, |
| "learning_rate": 0.00011768708188525956, |
| "loss": 8.266552734375, |
| "step": 9800 |
| }, |
| { |
| "epoch": 0.66, |
| "grad_norm": 1.0247814655303955, |
| "learning_rate": 0.00011368890270446406, |
| "loss": 8.150130004882813, |
| "step": 9900 |
| }, |
| { |
| "epoch": 0.6666666666666666, |
| "grad_norm": 1.0046054124832153, |
| "learning_rate": 0.00010973267227754613, |
| "loss": 8.194937744140624, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.6666666666666666, |
| "eval_accuracy": 0.30975342465753425, |
| "eval_loss": 8.204855918884277, |
| "eval_runtime": 65.6714, |
| "eval_samples_per_second": 7.614, |
| "eval_steps_per_second": 0.381, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.6733333333333333, |
| "grad_norm": 0.9764471650123596, |
| "learning_rate": 0.00010582031340402582, |
| "loss": 8.11692138671875, |
| "step": 10100 |
| }, |
| { |
| "epoch": 0.68, |
| "grad_norm": 1.0789597034454346, |
| "learning_rate": 0.00010195372756105512, |
| "loss": 8.137109375, |
| "step": 10200 |
| }, |
| { |
| "epoch": 0.6866666666666666, |
| "grad_norm": 1.0786553621292114, |
| "learning_rate": 9.813479397926537e-05, |
| "loss": 8.17680419921875, |
| "step": 10300 |
| }, |
| { |
| "epoch": 0.6933333333333334, |
| "grad_norm": 1.3332428932189941, |
| "learning_rate": 9.436536872942766e-05, |
| "loss": 8.15776123046875, |
| "step": 10400 |
| }, |
| { |
| "epoch": 0.7, |
| "grad_norm": 1.0370911359786987, |
| "learning_rate": 9.064728382036833e-05, |
| "loss": 8.137608642578124, |
| "step": 10500 |
| }, |
| { |
| "epoch": 0.7066666666666667, |
| "grad_norm": 1.081002950668335, |
| "learning_rate": 8.698234630857997e-05, |
| "loss": 8.0797705078125, |
| "step": 10600 |
| }, |
| { |
| "epoch": 0.7133333333333334, |
| "grad_norm": 1.0866549015045166, |
| "learning_rate": 8.337233741995907e-05, |
| "loss": 8.10765625, |
| "step": 10700 |
| }, |
| { |
| "epoch": 0.72, |
| "grad_norm": 1.4178708791732788, |
| "learning_rate": 7.9819011684098e-05, |
| "loss": 8.233140258789062, |
| "step": 10800 |
| }, |
| { |
| "epoch": 0.7266666666666667, |
| "grad_norm": 1.0158106088638306, |
| "learning_rate": 7.632409608155228e-05, |
| "loss": 8.12737060546875, |
| "step": 10900 |
| }, |
| { |
| "epoch": 0.7333333333333333, |
| "grad_norm": 1.0095257759094238, |
| "learning_rate": 7.288928920449636e-05, |
| "loss": 8.0778955078125, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.7333333333333333, |
| "eval_accuracy": 0.3137651663405088, |
| "eval_loss": 8.122584342956543, |
| "eval_runtime": 65.3252, |
| "eval_samples_per_second": 7.654, |
| "eval_steps_per_second": 0.383, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.74, |
| "grad_norm": 1.1707996129989624, |
| "learning_rate": 6.951626043117705e-05, |
| "loss": 8.111912841796874, |
| "step": 11100 |
| }, |
| { |
| "epoch": 0.7466666666666667, |
| "grad_norm": 1.1220096349716187, |
| "learning_rate": 6.620664911456616e-05, |
| "loss": 8.153399047851563, |
| "step": 11200 |
| }, |
| { |
| "epoch": 0.7533333333333333, |
| "grad_norm": 1.053057312965393, |
| "learning_rate": 6.296206378560454e-05, |
| "loss": 8.083673095703125, |
| "step": 11300 |
| }, |
| { |
| "epoch": 0.76, |
| "grad_norm": 1.0960450172424316, |
| "learning_rate": 5.978408137142759e-05, |
| "loss": 8.193185424804687, |
| "step": 11400 |
| }, |
| { |
| "epoch": 0.7666666666666667, |
| "grad_norm": 1.1686227321624756, |
| "learning_rate": 5.667424642894974e-05, |
| "loss": 8.13611572265625, |
| "step": 11500 |
| }, |
| { |
| "epoch": 0.7733333333333333, |
| "grad_norm": 1.103230595588684, |
| "learning_rate": 5.363407039418178e-05, |
| "loss": 8.102085571289063, |
| "step": 11600 |
| }, |
| { |
| "epoch": 0.78, |
| "grad_norm": 1.3868330717086792, |
| "learning_rate": 5.0665030847646266e-05, |
| "loss": 8.002344970703126, |
| "step": 11700 |
| }, |
| { |
| "epoch": 0.7866666666666666, |
| "grad_norm": 1.0746208429336548, |
| "learning_rate": 4.776857079624599e-05, |
| "loss": 8.036705322265625, |
| "step": 11800 |
| }, |
| { |
| "epoch": 0.7933333333333333, |
| "grad_norm": 1.161407470703125, |
| "learning_rate": 4.494609797193681e-05, |
| "loss": 7.990054931640625, |
| "step": 11900 |
| }, |
| { |
| "epoch": 0.8, |
| "grad_norm": 1.1083136796951294, |
| "learning_rate": 4.219898414754464e-05, |
| "loss": 8.060489501953125, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.8, |
| "eval_accuracy": 0.3169706457925636, |
| "eval_loss": 8.059541702270508, |
| "eval_runtime": 65.8107, |
| "eval_samples_per_second": 7.598, |
| "eval_steps_per_second": 0.38, |
| "step": 12000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 15000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|