| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 1000, |
| "global_step": 25000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": false, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.004, |
| "grad_norm": 0.6785533428192139, |
| "learning_rate": 0.003999874817760761, |
| "loss": 67.455439453125, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.7726423144340515, |
| "learning_rate": 0.00399943549159807, |
| "loss": 60.8977392578125, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.5963970422744751, |
| "learning_rate": 0.003998680178657526, |
| "loss": 58.8469140625, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.5085615515708923, |
| "learning_rate": 0.003997608998307273, |
| "loss": 57.544306640625, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.5642688870429993, |
| "learning_rate": 0.003996222119834506, |
| "loss": 56.976044921875, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.47304418683052063, |
| "learning_rate": 0.003994519762418718, |
| "loss": 56.1641748046875, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.45104825496673584, |
| "learning_rate": 0.003992502195097064, |
| "loss": 55.4853173828125, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.5336799621582031, |
| "learning_rate": 0.00399016973672184, |
| "loss": 54.9941357421875, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.5774248242378235, |
| "learning_rate": 0.0039875227559100935, |
| "loss": 54.54736328125, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.7312104105949402, |
| "learning_rate": 0.003984561670985366, |
| "loss": 54.2763134765625, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.1520635386119257, |
| "eval_loss": 53.98247146606445, |
| "eval_runtime": 23.3511, |
| "eval_samples_per_second": 5.353, |
| "eval_steps_per_second": 0.171, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.6063086986541748, |
| "learning_rate": 0.003981286949911586, |
| "loss": 53.98443359375, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.39948800206184387, |
| "learning_rate": 0.003977699110219106, |
| "loss": 53.585517578125, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.5279461145401001, |
| "learning_rate": 0.0039737987189229235, |
| "loss": 53.4054736328125, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.7522546648979187, |
| "learning_rate": 0.00396958639243306, |
| "loss": 53.0430322265625, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.7735885381698608, |
| "learning_rate": 0.003965062796457152, |
| "loss": 52.7428564453125, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 0.7539300322532654, |
| "learning_rate": 0.0039602286458952415, |
| "loss": 52.66658203125, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 0.49980393052101135, |
| "learning_rate": 0.0039550847047267945, |
| "loss": 52.561875, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 0.8233217000961304, |
| "learning_rate": 0.0039496317858899645, |
| "loss": 52.1762646484375, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 0.5626161098480225, |
| "learning_rate": 0.003943870751153116, |
| "loss": 51.989736328125, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 0.6408643126487732, |
| "learning_rate": 0.003937802510978631, |
| "loss": 51.8978759765625, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.15603128054740958, |
| "eval_loss": 51.726192474365234, |
| "eval_runtime": 7.2542, |
| "eval_samples_per_second": 17.231, |
| "eval_steps_per_second": 0.551, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.084, |
| "grad_norm": 0.5317835807800293, |
| "learning_rate": 0.0039314280243790255, |
| "loss": 51.5766015625, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.088, |
| "grad_norm": 0.7677087187767029, |
| "learning_rate": 0.003924748298765388, |
| "loss": 51.459482421875, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.092, |
| "grad_norm": 0.8987520933151245, |
| "learning_rate": 0.003917764389788164, |
| "loss": 51.2251513671875, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.096, |
| "grad_norm": 0.8509472608566284, |
| "learning_rate": 0.003910477401170333, |
| "loss": 51.086318359375, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 1.0043303966522217, |
| "learning_rate": 0.003902888484532972, |
| "loss": 51.013447265625, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.104, |
| "grad_norm": 0.6577029228210449, |
| "learning_rate": 0.0038949988392132555, |
| "loss": 50.6555419921875, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.108, |
| "grad_norm": 0.9883126020431519, |
| "learning_rate": 0.0038868097120749187, |
| "loss": 50.6927490234375, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.112, |
| "grad_norm": 1.0359454154968262, |
| "learning_rate": 0.003878322397311201, |
| "loss": 50.8522021484375, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.116, |
| "grad_norm": 0.8272144198417664, |
| "learning_rate": 0.0038695382362403186, |
| "loss": 50.349677734375, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 0.7374610304832458, |
| "learning_rate": 0.0038604586170934807, |
| "loss": 49.932939453125, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.12, |
| "eval_accuracy": 0.16219354838709676, |
| "eval_loss": 50.10387420654297, |
| "eval_runtime": 7.1742, |
| "eval_samples_per_second": 17.424, |
| "eval_steps_per_second": 0.558, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.124, |
| "grad_norm": 1.0479457378387451, |
| "learning_rate": 0.0038510849747955015, |
| "loss": 49.958212890625, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 1.4167745113372803, |
| "learning_rate": 0.0038414187907380216, |
| "loss": 49.9526025390625, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.132, |
| "grad_norm": 1.264487385749817, |
| "learning_rate": 0.0038314615925453986, |
| "loss": 49.983779296875, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.136, |
| "grad_norm": 1.5543415546417236, |
| "learning_rate": 0.003821214953833277, |
| "loss": 49.699599609375, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 1.4869952201843262, |
| "learning_rate": 0.0038106804939599037, |
| "loss": 49.46021484375, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.144, |
| "grad_norm": 1.2286678552627563, |
| "learning_rate": 0.003799859877770204, |
| "loss": 49.3774853515625, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.148, |
| "grad_norm": 2.0684049129486084, |
| "learning_rate": 0.003788754815332674, |
| "loss": 48.850126953125, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.152, |
| "grad_norm": 2.0739598274230957, |
| "learning_rate": 0.003777367061669124, |
| "loss": 48.7242919921875, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.156, |
| "grad_norm": 2.031198740005493, |
| "learning_rate": 0.0037656984164773206, |
| "loss": 48.9626513671875, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 1.9467569589614868, |
| "learning_rate": 0.003753750723846564, |
| "loss": 48.71037109375, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.16, |
| "eval_accuracy": 0.1748435972629521, |
| "eval_loss": 48.56401824951172, |
| "eval_runtime": 7.0573, |
| "eval_samples_per_second": 17.712, |
| "eval_steps_per_second": 0.567, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.164, |
| "grad_norm": 2.0323381423950195, |
| "learning_rate": 0.0037415258719662517, |
| "loss": 48.445341796875, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.168, |
| "grad_norm": 2.1837387084960938, |
| "learning_rate": 0.003729025792827473, |
| "loss": 48.2243603515625, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.172, |
| "grad_norm": 3.023264169692993, |
| "learning_rate": 0.0037162524619176844, |
| "loss": 48.10587890625, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.176, |
| "grad_norm": 2.3792617321014404, |
| "learning_rate": 0.0037032078979085024, |
| "loss": 48.301064453125, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.18, |
| "grad_norm": 3.003861665725708, |
| "learning_rate": 0.0036898941623366784, |
| "loss": 47.9244140625, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.184, |
| "grad_norm": 2.263864517211914, |
| "learning_rate": 0.003676313359278299, |
| "loss": 47.8005126953125, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.188, |
| "grad_norm": 2.5262391567230225, |
| "learning_rate": 0.0036624676350162613, |
| "loss": 47.87388671875, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.192, |
| "grad_norm": 2.9222702980041504, |
| "learning_rate": 0.0036483591777010794, |
| "loss": 47.5597412109375, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.196, |
| "grad_norm": 2.4583094120025635, |
| "learning_rate": 0.0036339902170050703, |
| "loss": 47.228359375, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 2.630126953125, |
| "learning_rate": 0.0036193630237699843, |
| "loss": 47.68759765625, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.2, |
| "eval_accuracy": 0.1841290322580645, |
| "eval_loss": 47.29731750488281, |
| "eval_runtime": 7.2208, |
| "eval_samples_per_second": 17.311, |
| "eval_steps_per_second": 0.554, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.204, |
| "grad_norm": 2.726828098297119, |
| "learning_rate": 0.003604479909648125, |
| "loss": 47.3409326171875, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.208, |
| "grad_norm": 2.4532039165496826, |
| "learning_rate": 0.00358934322673702, |
| "loss": 46.9724267578125, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.212, |
| "grad_norm": 3.2446208000183105, |
| "learning_rate": 0.003573955367207699, |
| "loss": 47.0882275390625, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.216, |
| "grad_norm": 3.021744966506958, |
| "learning_rate": 0.003558318762926643, |
| "loss": 46.8798291015625, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.22, |
| "grad_norm": 3.569321870803833, |
| "learning_rate": 0.003542435885071452, |
| "loss": 46.89576171875, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.224, |
| "grad_norm": 2.7224721908569336, |
| "learning_rate": 0.0035263092437403123, |
| "loss": 46.96203125, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.228, |
| "grad_norm": 2.9060356616973877, |
| "learning_rate": 0.003509941387555298, |
| "loss": 46.441123046875, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.232, |
| "grad_norm": 3.4122211933135986, |
| "learning_rate": 0.0034933349032595963, |
| "loss": 46.44693359375, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.236, |
| "grad_norm": 2.924081802368164, |
| "learning_rate": 0.0034764924153087035, |
| "loss": 46.6074658203125, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.24, |
| "grad_norm": 2.3687922954559326, |
| "learning_rate": 0.003459416585455659, |
| "loss": 46.527548828125, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.24, |
| "eval_accuracy": 0.19418279569892474, |
| "eval_loss": 46.334205627441406, |
| "eval_runtime": 7.5021, |
| "eval_samples_per_second": 16.662, |
| "eval_steps_per_second": 0.533, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.244, |
| "grad_norm": 3.177504062652588, |
| "learning_rate": 0.0034421101123303897, |
| "loss": 46.474375, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.248, |
| "grad_norm": 3.0214405059814453, |
| "learning_rate": 0.0034245757310132244, |
| "loss": 46.3790576171875, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.252, |
| "grad_norm": 2.933534860610962, |
| "learning_rate": 0.003406816212602642, |
| "loss": 46.0579736328125, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.256, |
| "grad_norm": 2.5542778968811035, |
| "learning_rate": 0.003388834363777341, |
| "loss": 46.0082861328125, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.26, |
| "grad_norm": 3.1228818893432617, |
| "learning_rate": 0.0033706330263526692, |
| "loss": 46.218310546875, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.264, |
| "grad_norm": 3.096647262573242, |
| "learning_rate": 0.003352215076831515, |
| "loss": 46.06939453125, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.268, |
| "grad_norm": 3.2484421730041504, |
| "learning_rate": 0.0033335834259497076, |
| "loss": 46.158388671875, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.272, |
| "grad_norm": 3.035205841064453, |
| "learning_rate": 0.003314741018216011, |
| "loss": 45.9556396484375, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.276, |
| "grad_norm": 3.4140381813049316, |
| "learning_rate": 0.0032956908314467803, |
| "loss": 45.8059375, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.28, |
| "grad_norm": 3.5898549556732178, |
| "learning_rate": 0.003276435876295352, |
| "loss": 45.785087890625, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.28, |
| "eval_accuracy": 0.2004613880742913, |
| "eval_loss": 45.5724983215332, |
| "eval_runtime": 7.1637, |
| "eval_samples_per_second": 17.449, |
| "eval_steps_per_second": 0.558, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.284, |
| "grad_norm": 3.6645097732543945, |
| "learning_rate": 0.0032569791957762473, |
| "loss": 45.450126953125, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.288, |
| "grad_norm": 3.449202060699463, |
| "learning_rate": 0.003237323864784261, |
| "loss": 45.6815869140625, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.292, |
| "grad_norm": 3.7189524173736572, |
| "learning_rate": 0.0032174729896085105, |
| "loss": 45.68208984375, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.296, |
| "grad_norm": 3.6789443492889404, |
| "learning_rate": 0.0031974297074415237, |
| "loss": 45.42337890625, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.3, |
| "grad_norm": 4.174764633178711, |
| "learning_rate": 0.003177197185883444, |
| "loss": 45.540263671875, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.304, |
| "grad_norm": 3.5856988430023193, |
| "learning_rate": 0.0031567786224414272, |
| "loss": 45.4182177734375, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.308, |
| "grad_norm": 3.8331663608551025, |
| "learning_rate": 0.003136177244024319, |
| "loss": 45.5009228515625, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.312, |
| "grad_norm": 3.373854160308838, |
| "learning_rate": 0.003115396306432675, |
| "loss": 45.4340283203125, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.316, |
| "grad_norm": 3.256587028503418, |
| "learning_rate": 0.003094439093844223, |
| "loss": 45.136494140625, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.32, |
| "grad_norm": 3.5598511695861816, |
| "learning_rate": 0.0030733089182948376, |
| "loss": 45.0095849609375, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.32, |
| "eval_accuracy": 0.2007086999022483, |
| "eval_loss": 45.0313720703125, |
| "eval_runtime": 6.9806, |
| "eval_samples_per_second": 17.907, |
| "eval_steps_per_second": 0.573, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.324, |
| "grad_norm": 3.4113783836364746, |
| "learning_rate": 0.0030520091191551142, |
| "loss": 45.00322265625, |
| "step": 8100 |
| }, |
| { |
| "epoch": 0.328, |
| "grad_norm": 3.566800832748413, |
| "learning_rate": 0.003030543062602622, |
| "loss": 45.1212841796875, |
| "step": 8200 |
| }, |
| { |
| "epoch": 0.332, |
| "grad_norm": 3.9237873554229736, |
| "learning_rate": 0.003008914141089914, |
| "loss": 45.1877978515625, |
| "step": 8300 |
| }, |
| { |
| "epoch": 0.336, |
| "grad_norm": 4.103876113891602, |
| "learning_rate": 0.0029871257728083995, |
| "loss": 45.1942626953125, |
| "step": 8400 |
| }, |
| { |
| "epoch": 0.34, |
| "grad_norm": 3.259410858154297, |
| "learning_rate": 0.0029651814011481333, |
| "loss": 44.929365234375, |
| "step": 8500 |
| }, |
| { |
| "epoch": 0.344, |
| "grad_norm": 3.518616199493408, |
| "learning_rate": 0.002943084494153631, |
| "loss": 44.5111328125, |
| "step": 8600 |
| }, |
| { |
| "epoch": 0.348, |
| "grad_norm": 3.544551372528076, |
| "learning_rate": 0.002920838543975789, |
| "loss": 44.8691015625, |
| "step": 8700 |
| }, |
| { |
| "epoch": 0.352, |
| "grad_norm": 3.5976264476776123, |
| "learning_rate": 0.0028984470663199887, |
| "loss": 44.924892578125, |
| "step": 8800 |
| }, |
| { |
| "epoch": 0.356, |
| "grad_norm": 3.3831348419189453, |
| "learning_rate": 0.0028759135998904796, |
| "loss": 44.773828125, |
| "step": 8900 |
| }, |
| { |
| "epoch": 0.36, |
| "grad_norm": 3.3760013580322266, |
| "learning_rate": 0.002853241705831138, |
| "loss": 44.8477294921875, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.36, |
| "eval_accuracy": 0.2051945259042033, |
| "eval_loss": 44.57013702392578, |
| "eval_runtime": 7.2683, |
| "eval_samples_per_second": 17.198, |
| "eval_steps_per_second": 0.55, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.364, |
| "grad_norm": 2.7893764972686768, |
| "learning_rate": 0.0028304349671626613, |
| "loss": 44.7609765625, |
| "step": 9100 |
| }, |
| { |
| "epoch": 0.368, |
| "grad_norm": 3.8668603897094727, |
| "learning_rate": 0.002807496988216319, |
| "loss": 44.6481201171875, |
| "step": 9200 |
| }, |
| { |
| "epoch": 0.372, |
| "grad_norm": 3.4443302154541016, |
| "learning_rate": 0.002784431394064333, |
| "loss": 44.4382373046875, |
| "step": 9300 |
| }, |
| { |
| "epoch": 0.376, |
| "grad_norm": 3.2611913681030273, |
| "learning_rate": 0.0027612418299469746, |
| "loss": 44.3794384765625, |
| "step": 9400 |
| }, |
| { |
| "epoch": 0.38, |
| "grad_norm": 2.5774736404418945, |
| "learning_rate": 0.0027379319606964806, |
| "loss": 44.4774462890625, |
| "step": 9500 |
| }, |
| { |
| "epoch": 0.384, |
| "grad_norm": 3.2907772064208984, |
| "learning_rate": 0.002714505470157871, |
| "loss": 44.4849267578125, |
| "step": 9600 |
| }, |
| { |
| "epoch": 0.388, |
| "grad_norm": 3.4475579261779785, |
| "learning_rate": 0.0026909660606067583, |
| "loss": 44.4125537109375, |
| "step": 9700 |
| }, |
| { |
| "epoch": 0.392, |
| "grad_norm": 4.186856269836426, |
| "learning_rate": 0.002667317452164251, |
| "loss": 44.404677734375, |
| "step": 9800 |
| }, |
| { |
| "epoch": 0.396, |
| "grad_norm": 3.1886963844299316, |
| "learning_rate": 0.0026435633822090316, |
| "loss": 44.2500146484375, |
| "step": 9900 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 3.3009133338928223, |
| "learning_rate": 0.002619707604786708, |
| "loss": 44.2538232421875, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.4, |
| "eval_accuracy": 0.20702737047898337, |
| "eval_loss": 44.112098693847656, |
| "eval_runtime": 7.3398, |
| "eval_samples_per_second": 17.03, |
| "eval_steps_per_second": 0.545, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.404, |
| "grad_norm": 3.5040667057037354, |
| "learning_rate": 0.0025957538900165333, |
| "loss": 44.2118505859375, |
| "step": 10100 |
| }, |
| { |
| "epoch": 0.408, |
| "grad_norm": 3.0154404640197754, |
| "learning_rate": 0.0025717060234955805, |
| "loss": 44.339365234375, |
| "step": 10200 |
| }, |
| { |
| "epoch": 0.412, |
| "grad_norm": 3.0024242401123047, |
| "learning_rate": 0.0025475678057004796, |
| "loss": 44.2085009765625, |
| "step": 10300 |
| }, |
| { |
| "epoch": 0.416, |
| "grad_norm": 2.868852138519287, |
| "learning_rate": 0.002523343051386793, |
| "loss": 44.1264453125, |
| "step": 10400 |
| }, |
| { |
| "epoch": 0.42, |
| "grad_norm": 3.098557472229004, |
| "learning_rate": 0.0024990355889861417, |
| "loss": 44.0060986328125, |
| "step": 10500 |
| }, |
| { |
| "epoch": 0.424, |
| "grad_norm": 2.94388747215271, |
| "learning_rate": 0.0024746492600011666, |
| "loss": 44.090751953125, |
| "step": 10600 |
| }, |
| { |
| "epoch": 0.428, |
| "grad_norm": 3.172335147857666, |
| "learning_rate": 0.002450187918398426, |
| "loss": 44.108330078125, |
| "step": 10700 |
| }, |
| { |
| "epoch": 0.432, |
| "grad_norm": 3.123621702194214, |
| "learning_rate": 0.0024256554299993214, |
| "loss": 44.00998046875, |
| "step": 10800 |
| }, |
| { |
| "epoch": 0.436, |
| "grad_norm": 2.8541388511657715, |
| "learning_rate": 0.002401055671869152, |
| "loss": 43.941904296875, |
| "step": 10900 |
| }, |
| { |
| "epoch": 0.44, |
| "grad_norm": 2.895085573196411, |
| "learning_rate": 0.0023763925317043903, |
| "loss": 43.792197265625, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.44, |
| "eval_accuracy": 0.20995894428152492, |
| "eval_loss": 43.750606536865234, |
| "eval_runtime": 6.9444, |
| "eval_samples_per_second": 18.0, |
| "eval_steps_per_second": 0.576, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.444, |
| "grad_norm": 2.346236228942871, |
| "learning_rate": 0.0023516699072182786, |
| "loss": 43.9089990234375, |
| "step": 11100 |
| }, |
| { |
| "epoch": 0.448, |
| "grad_norm": 2.3568971157073975, |
| "learning_rate": 0.002326891705524841, |
| "loss": 43.76560546875, |
| "step": 11200 |
| }, |
| { |
| "epoch": 0.452, |
| "grad_norm": 3.0081124305725098, |
| "learning_rate": 0.0023020618425214144, |
| "loss": 43.808974609375, |
| "step": 11300 |
| }, |
| { |
| "epoch": 0.456, |
| "grad_norm": 3.7043468952178955, |
| "learning_rate": 0.0022771842422697835, |
| "loss": 43.849228515625, |
| "step": 11400 |
| }, |
| { |
| "epoch": 0.46, |
| "grad_norm": 2.2964367866516113, |
| "learning_rate": 0.0022522628363760323, |
| "loss": 43.599345703125, |
| "step": 11500 |
| }, |
| { |
| "epoch": 0.464, |
| "grad_norm": 2.747478485107422, |
| "learning_rate": 0.002227301563369202, |
| "loss": 43.42494140625, |
| "step": 11600 |
| }, |
| { |
| "epoch": 0.468, |
| "grad_norm": 2.878131866455078, |
| "learning_rate": 0.0022023043680788512, |
| "loss": 43.6928955078125, |
| "step": 11700 |
| }, |
| { |
| "epoch": 0.472, |
| "grad_norm": 2.552713632583618, |
| "learning_rate": 0.0021772752010116247, |
| "loss": 43.6564453125, |
| "step": 11800 |
| }, |
| { |
| "epoch": 0.476, |
| "grad_norm": 2.9993340969085693, |
| "learning_rate": 0.002152218017726923, |
| "loss": 43.607763671875, |
| "step": 11900 |
| }, |
| { |
| "epoch": 0.48, |
| "grad_norm": 2.911576747894287, |
| "learning_rate": 0.002127136778211772, |
| "loss": 43.8162890625, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.48, |
| "eval_accuracy": 0.21305474095796675, |
| "eval_loss": 43.4426155090332, |
| "eval_runtime": 7.5722, |
| "eval_samples_per_second": 16.508, |
| "eval_steps_per_second": 0.528, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.484, |
| "grad_norm": 3.012925148010254, |
| "learning_rate": 0.0021020354462549986, |
| "loss": 43.3377099609375, |
| "step": 12100 |
| }, |
| { |
| "epoch": 0.488, |
| "grad_norm": 3.428558826446533, |
| "learning_rate": 0.0020769179888207967, |
| "loss": 43.5579638671875, |
| "step": 12200 |
| }, |
| { |
| "epoch": 0.492, |
| "grad_norm": 3.1122782230377197, |
| "learning_rate": 0.0020517883754218, |
| "loss": 43.4773828125, |
| "step": 12300 |
| }, |
| { |
| "epoch": 0.496, |
| "grad_norm": 2.5048506259918213, |
| "learning_rate": 0.0020266505774917446, |
| "loss": 43.3037744140625, |
| "step": 12400 |
| }, |
| { |
| "epoch": 0.5, |
| "grad_norm": 2.638044595718384, |
| "learning_rate": 0.0020015085677578355, |
| "loss": 43.562880859375, |
| "step": 12500 |
| }, |
| { |
| "epoch": 0.504, |
| "grad_norm": 2.780961036682129, |
| "learning_rate": 0.0019763663196129006, |
| "loss": 43.3517138671875, |
| "step": 12600 |
| }, |
| { |
| "epoch": 0.508, |
| "grad_norm": 1.9473482370376587, |
| "learning_rate": 0.0019512278064874483, |
| "loss": 43.109091796875, |
| "step": 12700 |
| }, |
| { |
| "epoch": 0.512, |
| "grad_norm": 3.174790143966675, |
| "learning_rate": 0.0019260970012217105, |
| "loss": 43.475107421875, |
| "step": 12800 |
| }, |
| { |
| "epoch": 0.516, |
| "grad_norm": 2.280482769012451, |
| "learning_rate": 0.001900977875437784, |
| "loss": 43.4238671875, |
| "step": 12900 |
| }, |
| { |
| "epoch": 0.52, |
| "grad_norm": 3.096158266067505, |
| "learning_rate": 0.0018758743989119647, |
| "loss": 43.194453125, |
| "step": 13000 |
| }, |
| { |
| "epoch": 0.52, |
| "eval_accuracy": 0.21564809384164224, |
| "eval_loss": 43.14116668701172, |
| "eval_runtime": 6.8523, |
| "eval_samples_per_second": 18.242, |
| "eval_steps_per_second": 0.584, |
| "step": 13000 |
| }, |
| { |
| "epoch": 0.524, |
| "grad_norm": 2.7369933128356934, |
| "learning_rate": 0.0018507905389473706, |
| "loss": 43.233857421875, |
| "step": 13100 |
| }, |
| { |
| "epoch": 0.528, |
| "grad_norm": 2.2296907901763916, |
| "learning_rate": 0.001825730259746958, |
| "loss": 43.15728515625, |
| "step": 13200 |
| }, |
| { |
| "epoch": 0.532, |
| "grad_norm": 3.018181324005127, |
| "learning_rate": 0.0018006975217870259, |
| "loss": 43.116005859375, |
| "step": 13300 |
| }, |
| { |
| "epoch": 0.536, |
| "grad_norm": 2.055466890335083, |
| "learning_rate": 0.0017756962811913122, |
| "loss": 43.1440869140625, |
| "step": 13400 |
| }, |
| { |
| "epoch": 0.54, |
| "grad_norm": 2.277271032333374, |
| "learning_rate": 0.0017507304891057722, |
| "loss": 43.4723193359375, |
| "step": 13500 |
| }, |
| { |
| "epoch": 0.544, |
| "grad_norm": 2.3194973468780518, |
| "learning_rate": 0.001725804091074152, |
| "loss": 43.233828125, |
| "step": 13600 |
| }, |
| { |
| "epoch": 0.548, |
| "grad_norm": 2.0250515937805176, |
| "learning_rate": 0.0017009210264144388, |
| "loss": 43.008115234375, |
| "step": 13700 |
| }, |
| { |
| "epoch": 0.552, |
| "grad_norm": 2.4250712394714355, |
| "learning_rate": 0.0016760852275963015, |
| "loss": 42.6800048828125, |
| "step": 13800 |
| }, |
| { |
| "epoch": 0.556, |
| "grad_norm": 1.7238541841506958, |
| "learning_rate": 0.0016513006196196098, |
| "loss": 42.93923828125, |
| "step": 13900 |
| }, |
| { |
| "epoch": 0.56, |
| "grad_norm": 1.7167351245880127, |
| "learning_rate": 0.0016265711193941361, |
| "loss": 43.267294921875, |
| "step": 14000 |
| }, |
| { |
| "epoch": 0.56, |
| "eval_accuracy": 0.2188279569892473, |
| "eval_loss": 42.8851318359375, |
| "eval_runtime": 7.0878, |
| "eval_samples_per_second": 17.636, |
| "eval_steps_per_second": 0.564, |
| "step": 14000 |
| }, |
| { |
| "epoch": 0.564, |
| "grad_norm": 2.029331684112549, |
| "learning_rate": 0.0016019006351205318, |
| "loss": 43.2043701171875, |
| "step": 14100 |
| }, |
| { |
| "epoch": 0.568, |
| "grad_norm": 2.470975399017334, |
| "learning_rate": 0.0015772930656726884, |
| "loss": 43.1994921875, |
| "step": 14200 |
| }, |
| { |
| "epoch": 0.572, |
| "grad_norm": 1.9896589517593384, |
| "learning_rate": 0.0015527522999815634, |
| "loss": 42.8731396484375, |
| "step": 14300 |
| }, |
| { |
| "epoch": 0.576, |
| "grad_norm": 2.9005286693573, |
| "learning_rate": 0.001528282216420582, |
| "loss": 42.8190380859375, |
| "step": 14400 |
| }, |
| { |
| "epoch": 0.58, |
| "grad_norm": 2.300215244293213, |
| "learning_rate": 0.0015038866821927073, |
| "loss": 42.9506982421875, |
| "step": 14500 |
| }, |
| { |
| "epoch": 0.584, |
| "grad_norm": 1.927196979522705, |
| "learning_rate": 0.0014795695527192762, |
| "loss": 42.97125, |
| "step": 14600 |
| }, |
| { |
| "epoch": 0.588, |
| "grad_norm": 1.9869160652160645, |
| "learning_rate": 0.001455334671030695, |
| "loss": 43.0368505859375, |
| "step": 14700 |
| }, |
| { |
| "epoch": 0.592, |
| "grad_norm": 2.05895733833313, |
| "learning_rate": 0.0014311858671590934, |
| "loss": 42.886181640625, |
| "step": 14800 |
| }, |
| { |
| "epoch": 0.596, |
| "grad_norm": 2.5150656700134277, |
| "learning_rate": 0.0014071269575330373, |
| "loss": 42.9019677734375, |
| "step": 14900 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 2.3601467609405518, |
| "learning_rate": 0.0013831617443743852, |
| "loss": 42.6784130859375, |
| "step": 15000 |
| }, |
| { |
| "epoch": 0.6, |
| "eval_accuracy": 0.22081524926686216, |
| "eval_loss": 42.696502685546875, |
| "eval_runtime": 7.2547, |
| "eval_samples_per_second": 17.23, |
| "eval_steps_per_second": 0.551, |
| "step": 15000 |
| }, |
| { |
| "epoch": 0.604, |
| "grad_norm": 2.1606075763702393, |
| "learning_rate": 0.0013592940150973943, |
| "loss": 42.9130859375, |
| "step": 15100 |
| }, |
| { |
| "epoch": 0.608, |
| "grad_norm": 2.021022319793701, |
| "learning_rate": 0.0013355275417101637, |
| "loss": 42.95048828125, |
| "step": 15200 |
| }, |
| { |
| "epoch": 0.612, |
| "grad_norm": 2.6041364669799805, |
| "learning_rate": 0.0013118660802185161, |
| "loss": 42.6790380859375, |
| "step": 15300 |
| }, |
| { |
| "epoch": 0.616, |
| "grad_norm": 2.121518135070801, |
| "learning_rate": 0.0012883133700324026, |
| "loss": 42.72169921875, |
| "step": 15400 |
| }, |
| { |
| "epoch": 0.62, |
| "grad_norm": 2.0749258995056152, |
| "learning_rate": 0.0012648731333749373, |
| "loss": 42.6244970703125, |
| "step": 15500 |
| }, |
| { |
| "epoch": 0.624, |
| "grad_norm": 2.3933210372924805, |
| "learning_rate": 0.0012415490746941434, |
| "loss": 42.629208984375, |
| "step": 15600 |
| }, |
| { |
| "epoch": 0.628, |
| "grad_norm": 2.163004159927368, |
| "learning_rate": 0.0012183448800775084, |
| "loss": 42.9462548828125, |
| "step": 15700 |
| }, |
| { |
| "epoch": 0.632, |
| "grad_norm": 1.9806509017944336, |
| "learning_rate": 0.0011952642166694445, |
| "loss": 42.9928662109375, |
| "step": 15800 |
| }, |
| { |
| "epoch": 0.636, |
| "grad_norm": 2.2631726264953613, |
| "learning_rate": 0.0011723107320917396, |
| "loss": 42.7688671875, |
| "step": 15900 |
| }, |
| { |
| "epoch": 0.64, |
| "grad_norm": 2.2971158027648926, |
| "learning_rate": 0.0011494880538670915, |
| "loss": 42.37572265625, |
| "step": 16000 |
| }, |
| { |
| "epoch": 0.64, |
| "eval_accuracy": 0.2221368523949169, |
| "eval_loss": 42.538299560546875, |
| "eval_runtime": 7.2739, |
| "eval_samples_per_second": 17.185, |
| "eval_steps_per_second": 0.55, |
| "step": 16000 |
| }, |
| { |
| "epoch": 0.644, |
| "grad_norm": 1.2762759923934937, |
| "learning_rate": 0.001126799788845827, |
| "loss": 42.4043359375, |
| "step": 16100 |
| }, |
| { |
| "epoch": 0.648, |
| "grad_norm": 1.4422513246536255, |
| "learning_rate": 0.0011042495226358784, |
| "loss": 43.033349609375, |
| "step": 16200 |
| }, |
| { |
| "epoch": 0.652, |
| "grad_norm": 1.816651463508606, |
| "learning_rate": 0.0010818408190361227, |
| "loss": 42.873017578125, |
| "step": 16300 |
| }, |
| { |
| "epoch": 0.656, |
| "grad_norm": 2.664728879928589, |
| "learning_rate": 0.0010595772194731657, |
| "loss": 42.72642578125, |
| "step": 16400 |
| }, |
| { |
| "epoch": 0.66, |
| "grad_norm": 2.1422553062438965, |
| "learning_rate": 0.0010374622424416619, |
| "loss": 42.725224609375, |
| "step": 16500 |
| }, |
| { |
| "epoch": 0.664, |
| "grad_norm": 2.044825315475464, |
| "learning_rate": 0.0010154993829482593, |
| "loss": 42.5387841796875, |
| "step": 16600 |
| }, |
| { |
| "epoch": 0.668, |
| "grad_norm": 1.400777816772461, |
| "learning_rate": 0.0009936921119592535, |
| "loss": 42.594912109375, |
| "step": 16700 |
| }, |
| { |
| "epoch": 0.672, |
| "grad_norm": 1.5564935207366943, |
| "learning_rate": 0.0009720438758520462, |
| "loss": 42.6428369140625, |
| "step": 16800 |
| }, |
| { |
| "epoch": 0.676, |
| "grad_norm": 2.554816961288452, |
| "learning_rate": 0.0009505580958704852, |
| "loss": 42.628232421875, |
| "step": 16900 |
| }, |
| { |
| "epoch": 0.68, |
| "grad_norm": 1.2824437618255615, |
| "learning_rate": 0.0009292381675841768, |
| "loss": 42.5581103515625, |
| "step": 17000 |
| }, |
| { |
| "epoch": 0.68, |
| "eval_accuracy": 0.22342913000977518, |
| "eval_loss": 42.39310836791992, |
| "eval_runtime": 7.1782, |
| "eval_samples_per_second": 17.414, |
| "eval_steps_per_second": 0.557, |
| "step": 17000 |
| }, |
| { |
| "epoch": 0.684, |
| "grad_norm": 1.755340576171875, |
| "learning_rate": 0.0009080874603518585, |
| "loss": 42.472861328125, |
| "step": 17100 |
| }, |
| { |
| "epoch": 0.688, |
| "grad_norm": 1.1157244443893433, |
| "learning_rate": 0.0008871093167889121, |
| "loss": 42.5861279296875, |
| "step": 17200 |
| }, |
| { |
| "epoch": 0.692, |
| "grad_norm": 1.2528804540634155, |
| "learning_rate": 0.0008663070522391008, |
| "loss": 42.4558740234375, |
| "step": 17300 |
| }, |
| { |
| "epoch": 0.696, |
| "grad_norm": 1.7521710395812988, |
| "learning_rate": 0.0008456839542506229, |
| "loss": 42.712470703125, |
| "step": 17400 |
| }, |
| { |
| "epoch": 0.7, |
| "grad_norm": 1.830307960510254, |
| "learning_rate": 0.0008252432820565525, |
| "loss": 42.67123046875, |
| "step": 17500 |
| }, |
| { |
| "epoch": 0.704, |
| "grad_norm": 1.6736066341400146, |
| "learning_rate": 0.0008049882660597558, |
| "loss": 42.42759765625, |
| "step": 17600 |
| }, |
| { |
| "epoch": 0.708, |
| "grad_norm": 1.5861881971359253, |
| "learning_rate": 0.0007849221073223665, |
| "loss": 42.5102734375, |
| "step": 17700 |
| }, |
| { |
| "epoch": 0.712, |
| "grad_norm": 1.659502387046814, |
| "learning_rate": 0.0007650479770598948, |
| "loss": 42.2067578125, |
| "step": 17800 |
| }, |
| { |
| "epoch": 0.716, |
| "grad_norm": 1.2964304685592651, |
| "learning_rate": 0.0007453690161400562, |
| "loss": 42.520947265625, |
| "step": 17900 |
| }, |
| { |
| "epoch": 0.72, |
| "grad_norm": 1.7349175214767456, |
| "learning_rate": 0.000725888334586394, |
| "loss": 42.7132568359375, |
| "step": 18000 |
| }, |
| { |
| "epoch": 0.72, |
| "eval_accuracy": 0.224316715542522, |
| "eval_loss": 42.29633331298828, |
| "eval_runtime": 7.2524, |
| "eval_samples_per_second": 17.236, |
| "eval_steps_per_second": 0.552, |
| "step": 18000 |
| }, |
| { |
| "epoch": 0.724, |
| "grad_norm": 1.7778606414794922, |
| "learning_rate": 0.0007066090110867782, |
| "loss": 42.758095703125, |
| "step": 18100 |
| }, |
| { |
| "epoch": 0.728, |
| "grad_norm": 1.6367878913879395, |
| "learning_rate": 0.0006875340925068554, |
| "loss": 42.4748291015625, |
| "step": 18200 |
| }, |
| { |
| "epoch": 0.732, |
| "grad_norm": 1.4886348247528076, |
| "learning_rate": 0.0006686665934085281, |
| "loss": 42.22041015625, |
| "step": 18300 |
| }, |
| { |
| "epoch": 0.736, |
| "grad_norm": 1.0366780757904053, |
| "learning_rate": 0.0006500094955735401, |
| "loss": 42.1452880859375, |
| "step": 18400 |
| }, |
| { |
| "epoch": 0.74, |
| "grad_norm": 1.189237117767334, |
| "learning_rate": 0.0006315657475322411, |
| "loss": 42.422060546875, |
| "step": 18500 |
| }, |
| { |
| "epoch": 0.744, |
| "grad_norm": 1.5588538646697998, |
| "learning_rate": 0.0006133382640976061, |
| "loss": 42.59294921875, |
| "step": 18600 |
| }, |
| { |
| "epoch": 0.748, |
| "grad_norm": 1.2961610555648804, |
| "learning_rate": 0.0005953299259045865, |
| "loss": 42.5683740234375, |
| "step": 18700 |
| }, |
| { |
| "epoch": 0.752, |
| "grad_norm": 1.513845682144165, |
| "learning_rate": 0.000577543578954858, |
| "loss": 42.31896484375, |
| "step": 18800 |
| }, |
| { |
| "epoch": 0.756, |
| "grad_norm": 1.0572335720062256, |
| "learning_rate": 0.0005599820341670454, |
| "loss": 42.3233642578125, |
| "step": 18900 |
| }, |
| { |
| "epoch": 0.76, |
| "grad_norm": 1.4302053451538086, |
| "learning_rate": 0.0005426480669324906, |
| "loss": 42.2752880859375, |
| "step": 19000 |
| }, |
| { |
| "epoch": 0.76, |
| "eval_accuracy": 0.22522482893450635, |
| "eval_loss": 42.213226318359375, |
| "eval_runtime": 7.0911, |
| "eval_samples_per_second": 17.628, |
| "eval_steps_per_second": 0.564, |
| "step": 19000 |
| }, |
| { |
| "epoch": 0.764, |
| "grad_norm": 1.4170910120010376, |
| "learning_rate": 0.0005255444166766354, |
| "loss": 42.50333984375, |
| "step": 19100 |
| }, |
| { |
| "epoch": 0.768, |
| "grad_norm": 1.3892245292663574, |
| "learning_rate": 0.0005086737864260862, |
| "loss": 42.474443359375, |
| "step": 19200 |
| }, |
| { |
| "epoch": 0.772, |
| "grad_norm": 1.5735739469528198, |
| "learning_rate": 0.0004920388423814364, |
| "loss": 42.25490234375, |
| "step": 19300 |
| }, |
| { |
| "epoch": 0.776, |
| "grad_norm": 1.2609901428222656, |
| "learning_rate": 0.0004756422134959031, |
| "loss": 42.3439208984375, |
| "step": 19400 |
| }, |
| { |
| "epoch": 0.78, |
| "grad_norm": 1.1414227485656738, |
| "learning_rate": 0.00045948649105985397, |
| "loss": 42.31787109375, |
| "step": 19500 |
| }, |
| { |
| "epoch": 0.784, |
| "grad_norm": 1.301095962524414, |
| "learning_rate": 0.0004435742282912836, |
| "loss": 42.43177734375, |
| "step": 19600 |
| }, |
| { |
| "epoch": 0.788, |
| "grad_norm": 1.1665655374526978, |
| "learning_rate": 0.0004279079399323087, |
| "loss": 42.386767578125, |
| "step": 19700 |
| }, |
| { |
| "epoch": 0.792, |
| "grad_norm": 1.5682238340377808, |
| "learning_rate": 0.0004124901018517433, |
| "loss": 42.2175439453125, |
| "step": 19800 |
| }, |
| { |
| "epoch": 0.796, |
| "grad_norm": 1.259811520576477, |
| "learning_rate": 0.00039732315065381754, |
| "loss": 42.252421875, |
| "step": 19900 |
| }, |
| { |
| "epoch": 0.8, |
| "grad_norm": 0.8927878141403198, |
| "learning_rate": 0.00038240948329310156, |
| "loss": 42.2313232421875, |
| "step": 20000 |
| }, |
| { |
| "epoch": 0.8, |
| "eval_accuracy": 0.2259481915933529, |
| "eval_loss": 42.14704895019531, |
| "eval_runtime": 6.8564, |
| "eval_samples_per_second": 18.231, |
| "eval_steps_per_second": 0.583, |
| "step": 20000 |
| }, |
| { |
| "epoch": 0.804, |
| "grad_norm": 1.055684208869934, |
| "learning_rate": 0.00036775145669569497, |
| "loss": 42.2956591796875, |
| "step": 20100 |
| }, |
| { |
| "epoch": 0.808, |
| "grad_norm": 1.0924086570739746, |
| "learning_rate": 0.0003533513873867442, |
| "loss": 42.5350732421875, |
| "step": 20200 |
| }, |
| { |
| "epoch": 0.812, |
| "grad_norm": 1.0316708087921143, |
| "learning_rate": 0.0003392115511243421, |
| "loss": 42.4560107421875, |
| "step": 20300 |
| }, |
| { |
| "epoch": 0.816, |
| "grad_norm": 1.1133615970611572, |
| "learning_rate": 0.00032533418253987323, |
| "loss": 42.3101513671875, |
| "step": 20400 |
| }, |
| { |
| "epoch": 0.82, |
| "grad_norm": 1.103115200996399, |
| "learning_rate": 0.00031172147478485514, |
| "loss": 42.077275390625, |
| "step": 20500 |
| }, |
| { |
| "epoch": 0.824, |
| "grad_norm": 1.1964024305343628, |
| "learning_rate": 0.00029837557918433946, |
| "loss": 42.031953125, |
| "step": 20600 |
| }, |
| { |
| "epoch": 0.828, |
| "grad_norm": 1.4636120796203613, |
| "learning_rate": 0.00028529860489691903, |
| "loss": 42.072587890625, |
| "step": 20700 |
| }, |
| { |
| "epoch": 0.832, |
| "grad_norm": 1.1224356889724731, |
| "learning_rate": 0.00027249261858140186, |
| "loss": 42.538427734375, |
| "step": 20800 |
| }, |
| { |
| "epoch": 0.836, |
| "grad_norm": 0.7813518643379211, |
| "learning_rate": 0.00025995964407019923, |
| "loss": 42.4304736328125, |
| "step": 20900 |
| }, |
| { |
| "epoch": 0.84, |
| "grad_norm": 0.8456737399101257, |
| "learning_rate": 0.0002477016620494856, |
| "loss": 42.2731982421875, |
| "step": 21000 |
| }, |
| { |
| "epoch": 0.84, |
| "eval_accuracy": 0.22656598240469208, |
| "eval_loss": 42.09526824951172, |
| "eval_runtime": 7.2222, |
| "eval_samples_per_second": 17.308, |
| "eval_steps_per_second": 0.554, |
| "step": 21000 |
| }, |
| { |
| "epoch": 0.844, |
| "grad_norm": 1.4627004861831665, |
| "learning_rate": 0.00023572060974617082, |
| "loss": 42.2644970703125, |
| "step": 21100 |
| }, |
| { |
| "epoch": 0.848, |
| "grad_norm": 1.2561938762664795, |
| "learning_rate": 0.0002240183806217493, |
| "loss": 42.0703955078125, |
| "step": 21200 |
| }, |
| { |
| "epoch": 0.852, |
| "grad_norm": 1.2216907739639282, |
| "learning_rate": 0.00021259682407305847, |
| "loss": 42.2020556640625, |
| "step": 21300 |
| }, |
| { |
| "epoch": 0.856, |
| "grad_norm": 0.8914040923118591, |
| "learning_rate": 0.00020145774514000438, |
| "loss": 42.357060546875, |
| "step": 21400 |
| }, |
| { |
| "epoch": 0.86, |
| "grad_norm": 1.1476701498031616, |
| "learning_rate": 0.00019060290422029657, |
| "loss": 42.13251953125, |
| "step": 21500 |
| }, |
| { |
| "epoch": 0.864, |
| "grad_norm": 1.119214653968811, |
| "learning_rate": 0.00018003401679123887, |
| "loss": 42.2200732421875, |
| "step": 21600 |
| }, |
| { |
| "epoch": 0.868, |
| "grad_norm": 0.9182925820350647, |
| "learning_rate": 0.0001697527531386196, |
| "loss": 42.22140625, |
| "step": 21700 |
| }, |
| { |
| "epoch": 0.872, |
| "grad_norm": 1.2138001918792725, |
| "learning_rate": 0.00015976073809273973, |
| "loss": 42.254619140625, |
| "step": 21800 |
| }, |
| { |
| "epoch": 0.876, |
| "grad_norm": 1.321160078048706, |
| "learning_rate": 0.00015005955077163137, |
| "loss": 42.439716796875, |
| "step": 21900 |
| }, |
| { |
| "epoch": 0.88, |
| "grad_norm": 0.8333455920219421, |
| "learning_rate": 0.00014065072433149607, |
| "loss": 42.4203271484375, |
| "step": 22000 |
| }, |
| { |
| "epoch": 0.88, |
| "eval_accuracy": 0.2271329423264907, |
| "eval_loss": 42.06480407714844, |
| "eval_runtime": 6.8562, |
| "eval_samples_per_second": 18.232, |
| "eval_steps_per_second": 0.583, |
| "step": 22000 |
| }, |
| { |
| "epoch": 0.884, |
| "grad_norm": 1.0895456075668335, |
| "learning_rate": 0.0001315357457244073, |
| "loss": 41.9967822265625, |
| "step": 22100 |
| }, |
| { |
| "epoch": 0.888, |
| "grad_norm": 0.9787936210632324, |
| "learning_rate": 0.000122716055463316, |
| "loss": 41.919248046875, |
| "step": 22200 |
| }, |
| { |
| "epoch": 0.892, |
| "grad_norm": 0.8595688343048096, |
| "learning_rate": 0.00011419304739439529, |
| "loss": 42.098642578125, |
| "step": 22300 |
| }, |
| { |
| "epoch": 0.896, |
| "grad_norm": 0.8901764154434204, |
| "learning_rate": 0.00010596806847675766, |
| "loss": 42.3766357421875, |
| "step": 22400 |
| }, |
| { |
| "epoch": 0.9, |
| "grad_norm": 0.9945716261863708, |
| "learning_rate": 9.80424185695874e-05, |
| "loss": 42.459091796875, |
| "step": 22500 |
| }, |
| { |
| "epoch": 0.904, |
| "grad_norm": 0.9749753475189209, |
| "learning_rate": 9.041735022671139e-05, |
| "loss": 42.227529296875, |
| "step": 22600 |
| }, |
| { |
| "epoch": 0.908, |
| "grad_norm": 0.8896678686141968, |
| "learning_rate": 8.309406849864854e-05, |
| "loss": 41.9632080078125, |
| "step": 22700 |
| }, |
| { |
| "epoch": 0.912, |
| "grad_norm": 0.9081159830093384, |
| "learning_rate": 7.607373074216573e-05, |
| "loss": 41.939306640625, |
| "step": 22800 |
| }, |
| { |
| "epoch": 0.916, |
| "grad_norm": 0.8322077989578247, |
| "learning_rate": 6.935744643737252e-05, |
| "loss": 42.21763671875, |
| "step": 22900 |
| }, |
| { |
| "epoch": 0.92, |
| "grad_norm": 0.8767455816268921, |
| "learning_rate": 6.294627701237876e-05, |
| "loss": 42.1103955078125, |
| "step": 23000 |
| }, |
| { |
| "epoch": 0.92, |
| "eval_accuracy": 0.22741544477028347, |
| "eval_loss": 42.044464111328125, |
| "eval_runtime": 7.5593, |
| "eval_samples_per_second": 16.536, |
| "eval_steps_per_second": 0.529, |
| "step": 23000 |
| }, |
| { |
| "epoch": 0.924, |
| "grad_norm": 0.8490807414054871, |
| "learning_rate": 5.6841235675552106e-05, |
| "loss": 42.408046875, |
| "step": 23100 |
| }, |
| { |
| "epoch": 0.928, |
| "grad_norm": 0.8971438407897949, |
| "learning_rate": 5.1043287255389206e-05, |
| "loss": 42.293935546875, |
| "step": 23200 |
| }, |
| { |
| "epoch": 0.932, |
| "grad_norm": 0.8031958341598511, |
| "learning_rate": 4.555334804803857e-05, |
| "loss": 42.037109375, |
| "step": 23300 |
| }, |
| { |
| "epoch": 0.936, |
| "grad_norm": 0.9145307540893555, |
| "learning_rate": 4.03722856724893e-05, |
| "loss": 42.266640625, |
| "step": 23400 |
| }, |
| { |
| "epoch": 0.94, |
| "grad_norm": 1.105343222618103, |
| "learning_rate": 3.5500918933456085e-05, |
| "loss": 41.9810498046875, |
| "step": 23500 |
| }, |
| { |
| "epoch": 0.944, |
| "grad_norm": 1.2218998670578003, |
| "learning_rate": 3.094001769197474e-05, |
| "loss": 42.188876953125, |
| "step": 23600 |
| }, |
| { |
| "epoch": 0.948, |
| "grad_norm": 0.8797289729118347, |
| "learning_rate": 2.669030274373663e-05, |
| "loss": 42.359853515625, |
| "step": 23700 |
| }, |
| { |
| "epoch": 0.952, |
| "grad_norm": 0.8956351280212402, |
| "learning_rate": 2.2752445705175097e-05, |
| "loss": 42.1323095703125, |
| "step": 23800 |
| }, |
| { |
| "epoch": 0.956, |
| "grad_norm": 0.8726316690444946, |
| "learning_rate": 1.9127068907324184e-05, |
| "loss": 42.337939453125, |
| "step": 23900 |
| }, |
| { |
| "epoch": 0.96, |
| "grad_norm": 1.1114789247512817, |
| "learning_rate": 1.581474529746729e-05, |
| "loss": 42.27337890625, |
| "step": 24000 |
| }, |
| { |
| "epoch": 0.96, |
| "eval_accuracy": 0.22743890518084067, |
| "eval_loss": 42.037322998046875, |
| "eval_runtime": 7.0814, |
| "eval_samples_per_second": 17.652, |
| "eval_steps_per_second": 0.565, |
| "step": 24000 |
| }, |
| { |
| "epoch": 0.964, |
| "grad_norm": 1.4101399183273315, |
| "learning_rate": 1.281599834858893e-05, |
| "loss": 42.2501513671875, |
| "step": 24100 |
| }, |
| { |
| "epoch": 0.968, |
| "grad_norm": 1.015312910079956, |
| "learning_rate": 1.0131301976646912e-05, |
| "loss": 42.1179931640625, |
| "step": 24200 |
| }, |
| { |
| "epoch": 0.972, |
| "grad_norm": 0.8839344382286072, |
| "learning_rate": 7.761080465675363e-06, |
| "loss": 42.1094140625, |
| "step": 24300 |
| }, |
| { |
| "epoch": 0.976, |
| "grad_norm": 1.0040006637573242, |
| "learning_rate": 5.705708400731702e-06, |
| "loss": 42.0864111328125, |
| "step": 24400 |
| }, |
| { |
| "epoch": 0.98, |
| "grad_norm": 0.9537904262542725, |
| "learning_rate": 3.965510608697098e-06, |
| "loss": 42.1570068359375, |
| "step": 24500 |
| }, |
| { |
| "epoch": 0.984, |
| "grad_norm": 0.9448596239089966, |
| "learning_rate": 2.5407621069435395e-06, |
| "loss": 42.424609375, |
| "step": 24600 |
| }, |
| { |
| "epoch": 0.988, |
| "grad_norm": 1.1038907766342163, |
| "learning_rate": 1.4316880598685967e-06, |
| "loss": 42.390634765625, |
| "step": 24700 |
| }, |
| { |
| "epoch": 0.992, |
| "grad_norm": 0.9914430379867554, |
| "learning_rate": 6.3846374331189e-07, |
| "loss": 42.1272607421875, |
| "step": 24800 |
| }, |
| { |
| "epoch": 0.996, |
| "grad_norm": 0.954636812210083, |
| "learning_rate": 1.6121451685435774e-07, |
| "loss": 42.1366357421875, |
| "step": 24900 |
| }, |
| { |
| "epoch": 1.0, |
| "grad_norm": 1.122141718864441, |
| "learning_rate": 1.5804007658104523e-11, |
| "loss": 42.0115185546875, |
| "step": 25000 |
| }, |
| { |
| "epoch": 1.0, |
| "eval_accuracy": 0.22748093841642228, |
| "eval_loss": 42.03590393066406, |
| "eval_runtime": 6.9933, |
| "eval_samples_per_second": 17.874, |
| "eval_steps_per_second": 0.572, |
| "step": 25000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 25000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|