diff --git "a/trainer_state.json" "b/trainer_state.json" new file mode 100644--- /dev/null +++ "b/trainer_state.json" @@ -0,0 +1,7835 @@ +{ + "best_global_step": 91000, + "best_metric": 3.857421875, + "best_model_checkpoint": "checkpoints/sonar_mini_5e4/checkpoint-91000", + "epoch": 1.0, + "eval_steps": 1000, + "global_step": 100000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.001, + "grad_norm": 2.9780468940734863, + "learning_rate": 6.7437500000000004e-06, + "loss": 7.4228, + "step": 100 + }, + { + "epoch": 0.002, + "grad_norm": 0.274592787027359, + "learning_rate": 1.29875e-05, + "loss": 5.1812, + "step": 200 + }, + { + "epoch": 0.003, + "grad_norm": 0.12851963937282562, + "learning_rate": 1.923125e-05, + "loss": 4.6227, + "step": 300 + }, + { + "epoch": 0.004, + "grad_norm": 0.0915624350309372, + "learning_rate": 2.5475000000000003e-05, + "loss": 4.4829, + "step": 400 + }, + { + "epoch": 0.005, + "grad_norm": 0.08360039442777634, + "learning_rate": 3.1718749999999995e-05, + "loss": 4.4024, + "step": 500 + }, + { + "epoch": 0.006, + "grad_norm": 0.08585469424724579, + "learning_rate": 3.7962499999999994e-05, + "loss": 4.3443, + "step": 600 + }, + { + "epoch": 0.007, + "grad_norm": 0.08225593715906143, + "learning_rate": 4.4206249999999994e-05, + "loss": 4.3023, + "step": 700 + }, + { + "epoch": 0.008, + "grad_norm": 0.09392552822828293, + "learning_rate": 5.045e-05, + "loss": 4.2707, + "step": 800 + }, + { + "epoch": 0.009, + "grad_norm": 0.08779512345790863, + "learning_rate": 5.669375e-05, + "loss": 4.2444, + "step": 900 + }, + { + "epoch": 0.01, + "grad_norm": 0.09750315546989441, + "learning_rate": 6.293749999999999e-05, + "loss": 4.22, + "step": 1000 + }, + { + "epoch": 0.011, + "grad_norm": 0.08571433275938034, + "learning_rate": 6.918125e-05, + "loss": 4.2002, + "step": 1100 + }, + { + "epoch": 0.012, + "grad_norm": 0.09333999454975128, + "learning_rate": 7.542499999999999e-05, + "loss": 4.1821, + "step": 1200 + }, + { + "epoch": 0.013, + "grad_norm": 0.09406431764364243, + "learning_rate": 8.166875e-05, + "loss": 4.166, + "step": 1300 + }, + { + "epoch": 0.014, + "grad_norm": 0.08803971856832504, + "learning_rate": 8.791249999999999e-05, + "loss": 4.1483, + "step": 1400 + }, + { + "epoch": 0.015, + "grad_norm": 0.10886548459529877, + "learning_rate": 9.415625e-05, + "loss": 4.1324, + "step": 1500 + }, + { + "epoch": 0.016, + "grad_norm": 0.10178254544734955, + "learning_rate": 0.0001004, + "loss": 4.1196, + "step": 1600 + }, + { + "epoch": 0.017, + "grad_norm": 0.09580868482589722, + "learning_rate": 0.00010664375, + "loss": 4.107, + "step": 1700 + }, + { + "epoch": 0.018, + "grad_norm": 0.0989207997918129, + "learning_rate": 0.0001128875, + "loss": 4.0939, + "step": 1800 + }, + { + "epoch": 0.019, + "grad_norm": 0.08986181765794754, + "learning_rate": 0.00011913125, + "loss": 4.0821, + "step": 1900 + }, + { + "epoch": 0.02, + "grad_norm": 0.10257107019424438, + "learning_rate": 0.000125375, + "loss": 4.0721, + "step": 2000 + }, + { + "epoch": 0.02, + "eval_loss": 4.4296875, + "eval_runtime": 33.6861, + "eval_samples_per_second": 17876.479, + "eval_steps_per_second": 17.485, + "step": 2000 + }, + { + "epoch": 0.021, + "grad_norm": 0.10034991055727005, + "learning_rate": 0.00013161875, + "loss": 4.0629, + "step": 2100 + }, + { + "epoch": 0.022, + "grad_norm": 0.10683776438236237, + "learning_rate": 0.0001378625, + "loss": 4.0515, + "step": 2200 + }, + { + "epoch": 0.023, + "grad_norm": 0.10092870891094208, + "learning_rate": 0.00014410625, + "loss": 4.0428, + "step": 2300 + }, + { + "epoch": 0.024, + "grad_norm": 0.11656210571527481, + "learning_rate": 0.00015035, + "loss": 4.0305, + "step": 2400 + }, + { + "epoch": 0.025, + "grad_norm": 0.10047990083694458, + "learning_rate": 0.00015659375, + "loss": 4.0236, + "step": 2500 + }, + { + "epoch": 0.026, + "grad_norm": 0.10436904430389404, + "learning_rate": 0.0001628375, + "loss": 4.014, + "step": 2600 + }, + { + "epoch": 0.027, + "grad_norm": 0.09785196930170059, + "learning_rate": 0.00016908125000000002, + "loss": 4.0062, + "step": 2700 + }, + { + "epoch": 0.028, + "grad_norm": 0.10417915135622025, + "learning_rate": 0.000175325, + "loss": 3.9974, + "step": 2800 + }, + { + "epoch": 0.029, + "grad_norm": 0.10269144922494888, + "learning_rate": 0.00018156875, + "loss": 3.99, + "step": 2900 + }, + { + "epoch": 0.03, + "grad_norm": 0.10645649582147598, + "learning_rate": 0.0001878125, + "loss": 3.9836, + "step": 3000 + }, + { + "epoch": 0.03, + "eval_loss": 4.3515625, + "eval_runtime": 33.1738, + "eval_samples_per_second": 18152.539, + "eval_steps_per_second": 17.755, + "step": 3000 + }, + { + "epoch": 0.031, + "grad_norm": 0.11746224015951157, + "learning_rate": 0.00019405625000000002, + "loss": 3.9753, + "step": 3100 + }, + { + "epoch": 0.032, + "grad_norm": 0.10270912200212479, + "learning_rate": 0.00020030000000000002, + "loss": 3.9693, + "step": 3200 + }, + { + "epoch": 0.033, + "grad_norm": 0.10418165475130081, + "learning_rate": 0.00020654375, + "loss": 3.9633, + "step": 3300 + }, + { + "epoch": 0.034, + "grad_norm": 0.10649538040161133, + "learning_rate": 0.0002127875, + "loss": 3.9519, + "step": 3400 + }, + { + "epoch": 0.035, + "grad_norm": 0.11068189144134521, + "learning_rate": 0.00021903125000000001, + "loss": 3.9479, + "step": 3500 + }, + { + "epoch": 0.036, + "grad_norm": 0.11896345019340515, + "learning_rate": 0.00022527500000000001, + "loss": 3.9408, + "step": 3600 + }, + { + "epoch": 0.037, + "grad_norm": 0.09746250510215759, + "learning_rate": 0.00023151875000000004, + "loss": 3.9353, + "step": 3700 + }, + { + "epoch": 0.038, + "grad_norm": 0.15779097378253937, + "learning_rate": 0.00023776249999999999, + "loss": 3.9304, + "step": 3800 + }, + { + "epoch": 0.039, + "grad_norm": 0.10786330699920654, + "learning_rate": 0.00024400625, + "loss": 3.9227, + "step": 3900 + }, + { + "epoch": 0.04, + "grad_norm": 0.11500537395477295, + "learning_rate": 0.00025025, + "loss": 3.9166, + "step": 4000 + }, + { + "epoch": 0.04, + "eval_loss": 4.296875, + "eval_runtime": 33.1436, + "eval_samples_per_second": 18169.071, + "eval_steps_per_second": 17.771, + "step": 4000 + }, + { + "epoch": 0.041, + "grad_norm": 0.13655492663383484, + "learning_rate": 0.00025649375, + "loss": 3.9107, + "step": 4100 + }, + { + "epoch": 0.042, + "grad_norm": 0.10026777535676956, + "learning_rate": 0.00026273750000000004, + "loss": 3.9054, + "step": 4200 + }, + { + "epoch": 0.043, + "grad_norm": 0.12338188290596008, + "learning_rate": 0.00026898125000000004, + "loss": 3.9006, + "step": 4300 + }, + { + "epoch": 0.044, + "grad_norm": 0.10768528282642365, + "learning_rate": 0.000275225, + "loss": 3.894, + "step": 4400 + }, + { + "epoch": 0.045, + "grad_norm": 0.10258147865533829, + "learning_rate": 0.00028146875, + "loss": 3.8912, + "step": 4500 + }, + { + "epoch": 0.046, + "grad_norm": 0.17049455642700195, + "learning_rate": 0.0002877125, + "loss": 3.8824, + "step": 4600 + }, + { + "epoch": 0.047, + "grad_norm": 0.10924670845270157, + "learning_rate": 0.00029395625000000003, + "loss": 3.8803, + "step": 4700 + }, + { + "epoch": 0.048, + "grad_norm": 0.12295184284448624, + "learning_rate": 0.0003002, + "loss": 3.8727, + "step": 4800 + }, + { + "epoch": 0.049, + "grad_norm": 0.1435551792383194, + "learning_rate": 0.00030644375000000003, + "loss": 3.8705, + "step": 4900 + }, + { + "epoch": 0.05, + "grad_norm": 0.16839157044887543, + "learning_rate": 0.00031268750000000003, + "loss": 3.8646, + "step": 5000 + }, + { + "epoch": 0.05, + "eval_loss": 4.25, + "eval_runtime": 33.1236, + "eval_samples_per_second": 18180.027, + "eval_steps_per_second": 17.782, + "step": 5000 + }, + { + "epoch": 0.051, + "grad_norm": 0.12637090682983398, + "learning_rate": 0.00031893125000000003, + "loss": 3.86, + "step": 5100 + }, + { + "epoch": 0.052, + "grad_norm": 0.16098865866661072, + "learning_rate": 0.000325175, + "loss": 3.8559, + "step": 5200 + }, + { + "epoch": 0.053, + "grad_norm": 0.10894348472356796, + "learning_rate": 0.00033141875, + "loss": 3.8512, + "step": 5300 + }, + { + "epoch": 0.054, + "grad_norm": 0.11462244391441345, + "learning_rate": 0.00033766250000000003, + "loss": 3.8469, + "step": 5400 + }, + { + "epoch": 0.055, + "grad_norm": 0.111247718334198, + "learning_rate": 0.00034390625000000003, + "loss": 3.842, + "step": 5500 + }, + { + "epoch": 0.056, + "grad_norm": 0.14288830757141113, + "learning_rate": 0.00035015, + "loss": 3.8373, + "step": 5600 + }, + { + "epoch": 0.057, + "grad_norm": 0.14327366650104523, + "learning_rate": 0.00035639375, + "loss": 3.8346, + "step": 5700 + }, + { + "epoch": 0.058, + "grad_norm": 0.10640984773635864, + "learning_rate": 0.0003626375, + "loss": 3.8318, + "step": 5800 + }, + { + "epoch": 0.059, + "grad_norm": 0.12228193134069443, + "learning_rate": 0.0003688812500000001, + "loss": 3.8274, + "step": 5900 + }, + { + "epoch": 0.06, + "grad_norm": 0.12095323950052261, + "learning_rate": 0.00037512499999999997, + "loss": 3.8233, + "step": 6000 + }, + { + "epoch": 0.06, + "eval_loss": 4.21875, + "eval_runtime": 33.1908, + "eval_samples_per_second": 18143.214, + "eval_steps_per_second": 17.746, + "step": 6000 + }, + { + "epoch": 0.061, + "grad_norm": 0.17427460849285126, + "learning_rate": 0.00038136874999999997, + "loss": 3.8206, + "step": 6100 + }, + { + "epoch": 0.062, + "grad_norm": 0.17270584404468536, + "learning_rate": 0.0003876125, + "loss": 3.8141, + "step": 6200 + }, + { + "epoch": 0.063, + "grad_norm": 0.10496606677770615, + "learning_rate": 0.00039385624999999997, + "loss": 3.8104, + "step": 6300 + }, + { + "epoch": 0.064, + "grad_norm": 0.18062132596969604, + "learning_rate": 0.0004001, + "loss": 3.8084, + "step": 6400 + }, + { + "epoch": 0.065, + "grad_norm": 0.1837543398141861, + "learning_rate": 0.00040634375, + "loss": 3.8042, + "step": 6500 + }, + { + "epoch": 0.066, + "grad_norm": 0.1967838704586029, + "learning_rate": 0.0004125875, + "loss": 3.7992, + "step": 6600 + }, + { + "epoch": 0.067, + "grad_norm": 0.23508614301681519, + "learning_rate": 0.00041883125, + "loss": 3.7988, + "step": 6700 + }, + { + "epoch": 0.068, + "grad_norm": 0.15716159343719482, + "learning_rate": 0.00042507499999999996, + "loss": 3.7925, + "step": 6800 + }, + { + "epoch": 0.069, + "grad_norm": 0.22673483192920685, + "learning_rate": 0.00043131875, + "loss": 3.7906, + "step": 6900 + }, + { + "epoch": 0.07, + "grad_norm": 0.2622189521789551, + "learning_rate": 0.0004375625, + "loss": 3.7868, + "step": 7000 + }, + { + "epoch": 0.07, + "eval_loss": 4.19921875, + "eval_runtime": 33.1108, + "eval_samples_per_second": 18187.079, + "eval_steps_per_second": 17.789, + "step": 7000 + }, + { + "epoch": 0.071, + "grad_norm": 0.12971626222133636, + "learning_rate": 0.00044380624999999996, + "loss": 3.7834, + "step": 7100 + }, + { + "epoch": 0.072, + "grad_norm": 0.12215161323547363, + "learning_rate": 0.00045005, + "loss": 3.7807, + "step": 7200 + }, + { + "epoch": 0.073, + "grad_norm": 0.17193715274333954, + "learning_rate": 0.00045629375, + "loss": 3.7767, + "step": 7300 + }, + { + "epoch": 0.074, + "grad_norm": 0.14749088883399963, + "learning_rate": 0.00046253750000000007, + "loss": 3.7741, + "step": 7400 + }, + { + "epoch": 0.075, + "grad_norm": 0.21286271512508392, + "learning_rate": 0.00046878125, + "loss": 3.771, + "step": 7500 + }, + { + "epoch": 0.076, + "grad_norm": 0.2201075553894043, + "learning_rate": 0.00047502499999999996, + "loss": 3.7676, + "step": 7600 + }, + { + "epoch": 0.077, + "grad_norm": 0.20476743578910828, + "learning_rate": 0.00048126875, + "loss": 3.7648, + "step": 7700 + }, + { + "epoch": 0.078, + "grad_norm": 0.1267070174217224, + "learning_rate": 0.0004875125, + "loss": 3.7634, + "step": 7800 + }, + { + "epoch": 0.079, + "grad_norm": 0.13982422649860382, + "learning_rate": 0.00049375625, + "loss": 3.7585, + "step": 7900 + }, + { + "epoch": 0.08, + "grad_norm": 0.17995713651180267, + "learning_rate": 0.0005, + "loss": 3.7567, + "step": 8000 + }, + { + "epoch": 0.08, + "eval_loss": 4.15625, + "eval_runtime": 33.1093, + "eval_samples_per_second": 18187.899, + "eval_steps_per_second": 17.79, + "step": 8000 + }, + { + "epoch": 0.081, + "grad_norm": 0.14613783359527588, + "learning_rate": 0.0004969039949999533, + "loss": 3.7543, + "step": 8100 + }, + { + "epoch": 0.082, + "grad_norm": 0.13322409987449646, + "learning_rate": 0.0004938647983247948, + "loss": 3.7495, + "step": 8200 + }, + { + "epoch": 0.083, + "grad_norm": 0.23195946216583252, + "learning_rate": 0.0004908806936738159, + "loss": 3.7483, + "step": 8300 + }, + { + "epoch": 0.084, + "grad_norm": 0.22853663563728333, + "learning_rate": 0.00048795003647426655, + "loss": 3.7421, + "step": 8400 + }, + { + "epoch": 0.085, + "grad_norm": 0.19258145987987518, + "learning_rate": 0.0004850712500726659, + "loss": 3.74, + "step": 8500 + }, + { + "epoch": 0.086, + "grad_norm": 0.15334059298038483, + "learning_rate": 0.0004822428221704122, + "loss": 3.7385, + "step": 8600 + }, + { + "epoch": 0.087, + "grad_norm": 0.13609085977077484, + "learning_rate": 0.0004794633014853842, + "loss": 3.7364, + "step": 8700 + }, + { + "epoch": 0.088, + "grad_norm": 0.16103780269622803, + "learning_rate": 0.00047673129462279614, + "loss": 3.7331, + "step": 8800 + }, + { + "epoch": 0.089, + "grad_norm": 0.13547872006893158, + "learning_rate": 0.00047404546313997725, + "loss": 3.7288, + "step": 8900 + }, + { + "epoch": 0.09, + "grad_norm": 0.1519702523946762, + "learning_rate": 0.00047140452079103175, + "loss": 3.7258, + "step": 9000 + }, + { + "epoch": 0.09, + "eval_loss": 4.13671875, + "eval_runtime": 33.1497, + "eval_samples_per_second": 18165.735, + "eval_steps_per_second": 17.768, + "step": 9000 + }, + { + "epoch": 0.091, + "grad_norm": 0.13400638103485107, + "learning_rate": 0.00046880723093849544, + "loss": 3.7264, + "step": 9100 + }, + { + "epoch": 0.092, + "grad_norm": 0.17761798202991486, + "learning_rate": 0.0004662524041201569, + "loss": 3.7228, + "step": 9200 + }, + { + "epoch": 0.093, + "grad_norm": 0.10788439959287643, + "learning_rate": 0.00046373889576016824, + "loss": 3.7187, + "step": 9300 + }, + { + "epoch": 0.094, + "grad_norm": 0.13771696388721466, + "learning_rate": 0.00046126560401444256, + "loss": 3.7147, + "step": 9400 + }, + { + "epoch": 0.095, + "grad_norm": 0.1324366182088852, + "learning_rate": 0.0004588314677411235, + "loss": 3.7148, + "step": 9500 + }, + { + "epoch": 0.096, + "grad_norm": 0.15091444551944733, + "learning_rate": 0.0004564354645876385, + "loss": 3.7135, + "step": 9600 + }, + { + "epoch": 0.097, + "grad_norm": 0.1656925231218338, + "learning_rate": 0.0004540766091864998, + "loss": 3.7098, + "step": 9700 + }, + { + "epoch": 0.098, + "grad_norm": 0.13978664577007294, + "learning_rate": 0.00045175395145262567, + "loss": 3.7062, + "step": 9800 + }, + { + "epoch": 0.099, + "grad_norm": 0.10810708999633789, + "learning_rate": 0.0004494665749754947, + "loss": 3.7057, + "step": 9900 + }, + { + "epoch": 0.1, + "grad_norm": 0.1327275186777115, + "learning_rate": 0.00044721359549995795, + "loss": 3.7022, + "step": 10000 + }, + { + "epoch": 0.1, + "eval_loss": 4.11328125, + "eval_runtime": 33.1707, + "eval_samples_per_second": 18154.235, + "eval_steps_per_second": 17.757, + "step": 10000 + }, + { + "epoch": 0.101, + "grad_norm": 0.1312229484319687, + "learning_rate": 0.0004449941594899848, + "loss": 3.7018, + "step": 10100 + }, + { + "epoch": 0.102, + "grad_norm": 0.12260957062244415, + "learning_rate": 0.0004428074427700477, + "loss": 3.6961, + "step": 10200 + }, + { + "epoch": 0.103, + "grad_norm": 0.10720884054899216, + "learning_rate": 0.0004406526492392317, + "loss": 3.6958, + "step": 10300 + }, + { + "epoch": 0.104, + "grad_norm": 0.10942485928535461, + "learning_rate": 0.0004385290096535146, + "loss": 3.6949, + "step": 10400 + }, + { + "epoch": 0.105, + "grad_norm": 0.11342720687389374, + "learning_rate": 0.0004364357804719848, + "loss": 3.6954, + "step": 10500 + }, + { + "epoch": 0.106, + "grad_norm": 0.12471634149551392, + "learning_rate": 0.0004343722427630694, + "loss": 3.6926, + "step": 10600 + }, + { + "epoch": 0.107, + "grad_norm": 0.11318652331829071, + "learning_rate": 0.00043233770116711697, + "loss": 3.6912, + "step": 10700 + }, + { + "epoch": 0.108, + "grad_norm": 0.11893333494663239, + "learning_rate": 0.0004303314829119352, + "loss": 3.6893, + "step": 10800 + }, + { + "epoch": 0.109, + "grad_norm": 0.12605136632919312, + "learning_rate": 0.0004283529368781193, + "loss": 3.6862, + "step": 10900 + }, + { + "epoch": 0.11, + "grad_norm": 0.24888132512569427, + "learning_rate": 0.00042640143271122083, + "loss": 3.6851, + "step": 11000 + }, + { + "epoch": 0.11, + "eval_loss": 4.09765625, + "eval_runtime": 33.0869, + "eval_samples_per_second": 18200.176, + "eval_steps_per_second": 17.802, + "step": 11000 + }, + { + "epoch": 0.111, + "grad_norm": 0.1126684695482254, + "learning_rate": 0.00042447635997800896, + "loss": 3.6798, + "step": 11100 + }, + { + "epoch": 0.112, + "grad_norm": 0.16141295433044434, + "learning_rate": 0.0004225771273642583, + "loss": 3.6806, + "step": 11200 + }, + { + "epoch": 0.113, + "grad_norm": 0.18238785862922668, + "learning_rate": 0.0004207031619116712, + "loss": 3.6773, + "step": 11300 + }, + { + "epoch": 0.114, + "grad_norm": 0.14396843314170837, + "learning_rate": 0.0004188539082916955, + "loss": 3.6765, + "step": 11400 + }, + { + "epoch": 0.115, + "grad_norm": 0.20065593719482422, + "learning_rate": 0.0004170288281141495, + "loss": 3.673, + "step": 11500 + }, + { + "epoch": 0.116, + "grad_norm": 0.12579910457134247, + "learning_rate": 0.00041522739926869986, + "loss": 3.6727, + "step": 11600 + }, + { + "epoch": 0.117, + "grad_norm": 0.10580720007419586, + "learning_rate": 0.0004134491152973616, + "loss": 3.6699, + "step": 11700 + }, + { + "epoch": 0.118, + "grad_norm": 0.12733343243598938, + "learning_rate": 0.00041169348479630914, + "loss": 3.6685, + "step": 11800 + }, + { + "epoch": 0.119, + "grad_norm": 0.09983912110328674, + "learning_rate": 0.0004099600308453939, + "loss": 3.6677, + "step": 11900 + }, + { + "epoch": 0.12, + "grad_norm": 0.1023598462343216, + "learning_rate": 0.0004082482904638631, + "loss": 3.6669, + "step": 12000 + }, + { + "epoch": 0.12, + "eval_loss": 4.08203125, + "eval_runtime": 33.0923, + "eval_samples_per_second": 18197.203, + "eval_steps_per_second": 17.799, + "step": 12000 + }, + { + "epoch": 0.121, + "grad_norm": 0.12581580877304077, + "learning_rate": 0.0004065578140908709, + "loss": 3.6641, + "step": 12100 + }, + { + "epoch": 0.122, + "grad_norm": 0.11037498712539673, + "learning_rate": 0.000404888165089458, + "loss": 3.664, + "step": 12200 + }, + { + "epoch": 0.123, + "grad_norm": 0.11614446341991425, + "learning_rate": 0.0004032389192727559, + "loss": 3.6598, + "step": 12300 + }, + { + "epoch": 0.124, + "grad_norm": 0.12279075384140015, + "learning_rate": 0.0004016096644512495, + "loss": 3.6603, + "step": 12400 + }, + { + "epoch": 0.125, + "grad_norm": 0.1049276739358902, + "learning_rate": 0.0004, + "loss": 3.6569, + "step": 12500 + }, + { + "epoch": 0.126, + "grad_norm": 0.16340377926826477, + "learning_rate": 0.0003984095364447979, + "loss": 3.6581, + "step": 12600 + }, + { + "epoch": 0.127, + "grad_norm": 0.10878612846136093, + "learning_rate": 0.00039683789506627255, + "loss": 3.6548, + "step": 12700 + }, + { + "epoch": 0.128, + "grad_norm": 0.11425680667161942, + "learning_rate": 0.00039528470752104737, + "loss": 3.6564, + "step": 12800 + }, + { + "epoch": 0.129, + "grad_norm": 0.12371216714382172, + "learning_rate": 0.0003937496154790789, + "loss": 3.6509, + "step": 12900 + }, + { + "epoch": 0.13, + "grad_norm": 0.1376781165599823, + "learning_rate": 0.0003922322702763681, + "loss": 3.6531, + "step": 13000 + }, + { + "epoch": 0.13, + "eval_loss": 4.0625, + "eval_runtime": 33.0799, + "eval_samples_per_second": 18204.035, + "eval_steps_per_second": 17.805, + "step": 13000 + }, + { + "epoch": 0.131, + "grad_norm": 0.15412083268165588, + "learning_rate": 0.00039073233258228173, + "loss": 3.6495, + "step": 13100 + }, + { + "epoch": 0.132, + "grad_norm": 0.10344259440898895, + "learning_rate": 0.0003892494720807615, + "loss": 3.652, + "step": 13200 + }, + { + "epoch": 0.133, + "grad_norm": 0.12037135660648346, + "learning_rate": 0.0003877833671647406, + "loss": 3.6492, + "step": 13300 + }, + { + "epoch": 0.134, + "grad_norm": 0.10815135389566422, + "learning_rate": 0.0003863337046431279, + "loss": 3.6461, + "step": 13400 + }, + { + "epoch": 0.135, + "grad_norm": 0.14830388128757477, + "learning_rate": 0.00038490017945975053, + "loss": 3.6432, + "step": 13500 + }, + { + "epoch": 0.136, + "grad_norm": 0.14780588448047638, + "learning_rate": 0.0003834824944236852, + "loss": 3.641, + "step": 13600 + }, + { + "epoch": 0.137, + "grad_norm": 0.13959766924381256, + "learning_rate": 0.00038208035995043505, + "loss": 3.6423, + "step": 13700 + }, + { + "epoch": 0.138, + "grad_norm": 0.1739835888147354, + "learning_rate": 0.0003806934938134405, + "loss": 3.6412, + "step": 13800 + }, + { + "epoch": 0.139, + "grad_norm": 0.1504901498556137, + "learning_rate": 0.0003793216209054408, + "loss": 3.6402, + "step": 13900 + }, + { + "epoch": 0.14, + "grad_norm": 0.11134779453277588, + "learning_rate": 0.0003779644730092272, + "loss": 3.6379, + "step": 14000 + }, + { + "epoch": 0.14, + "eval_loss": 4.0546875, + "eval_runtime": 33.0435, + "eval_samples_per_second": 18224.12, + "eval_steps_per_second": 17.825, + "step": 14000 + }, + { + "epoch": 0.141, + "grad_norm": 0.14684522151947021, + "learning_rate": 0.0003766217885773547, + "loss": 3.638, + "step": 14100 + }, + { + "epoch": 0.142, + "grad_norm": 0.10334658622741699, + "learning_rate": 0.0003752933125204008, + "loss": 3.6364, + "step": 14200 + }, + { + "epoch": 0.143, + "grad_norm": 0.1433073729276657, + "learning_rate": 0.00037397879600338285, + "loss": 3.636, + "step": 14300 + }, + { + "epoch": 0.144, + "grad_norm": 0.112115278840065, + "learning_rate": 0.000372677996249965, + "loss": 3.6345, + "step": 14400 + }, + { + "epoch": 0.145, + "grad_norm": 0.11925119906663895, + "learning_rate": 0.0003713906763541037, + "loss": 3.6319, + "step": 14500 + }, + { + "epoch": 0.146, + "grad_norm": 0.12104904651641846, + "learning_rate": 0.00037011660509880264, + "loss": 3.6304, + "step": 14600 + }, + { + "epoch": 0.147, + "grad_norm": 0.13862280547618866, + "learning_rate": 0.0003688555567816587, + "loss": 3.6297, + "step": 14700 + }, + { + "epoch": 0.148, + "grad_norm": 0.1608782410621643, + "learning_rate": 0.00036760731104690394, + "loss": 3.6296, + "step": 14800 + }, + { + "epoch": 0.149, + "grad_norm": 0.09898547828197479, + "learning_rate": 0.0003663716527236559, + "loss": 3.6265, + "step": 14900 + }, + { + "epoch": 0.15, + "grad_norm": 0.11533054709434509, + "learning_rate": 0.00036514837167011074, + "loss": 3.6266, + "step": 15000 + }, + { + "epoch": 0.15, + "eval_loss": 4.05078125, + "eval_runtime": 33.0573, + "eval_samples_per_second": 18216.467, + "eval_steps_per_second": 17.818, + "step": 15000 + }, + { + "epoch": 0.151, + "grad_norm": 0.10128284990787506, + "learning_rate": 0.0003639372626234195, + "loss": 3.6259, + "step": 15100 + }, + { + "epoch": 0.152, + "grad_norm": 0.12907330691814423, + "learning_rate": 0.0003627381250550059, + "loss": 3.6277, + "step": 15200 + }, + { + "epoch": 0.153, + "grad_norm": 0.09283538907766342, + "learning_rate": 0.0003615507630310936, + "loss": 3.6219, + "step": 15300 + }, + { + "epoch": 0.154, + "grad_norm": 0.10961660742759705, + "learning_rate": 0.0003603749850782236, + "loss": 3.6233, + "step": 15400 + }, + { + "epoch": 0.155, + "grad_norm": 0.11972799897193909, + "learning_rate": 0.00035921060405354985, + "loss": 3.6202, + "step": 15500 + }, + { + "epoch": 0.156, + "grad_norm": 0.11016960442066193, + "learning_rate": 0.0003580574370197164, + "loss": 3.6204, + "step": 15600 + }, + { + "epoch": 0.157, + "grad_norm": 0.11285857856273651, + "learning_rate": 0.0003569153051241248, + "loss": 3.6205, + "step": 15700 + }, + { + "epoch": 0.158, + "grad_norm": 0.11873608082532883, + "learning_rate": 0.00035578403348241, + "loss": 3.619, + "step": 15800 + }, + { + "epoch": 0.159, + "grad_norm": 0.1388588696718216, + "learning_rate": 0.0003546634510659543, + "loss": 3.6178, + "step": 15900 + }, + { + "epoch": 0.16, + "grad_norm": 0.10858651995658875, + "learning_rate": 0.00035355339059327376, + "loss": 3.6183, + "step": 16000 + }, + { + "epoch": 0.16, + "eval_loss": 4.03515625, + "eval_runtime": 33.0196, + "eval_samples_per_second": 18237.308, + "eval_steps_per_second": 17.838, + "step": 16000 + }, + { + "epoch": 0.161, + "grad_norm": 0.09829683601856232, + "learning_rate": 0.0003524536884251207, + "loss": 3.618, + "step": 16100 + }, + { + "epoch": 0.162, + "grad_norm": 0.12840525805950165, + "learning_rate": 0.00035136418446315325, + "loss": 3.6153, + "step": 16200 + }, + { + "epoch": 0.163, + "grad_norm": 0.13450373709201813, + "learning_rate": 0.00035028472205202936, + "loss": 3.6142, + "step": 16300 + }, + { + "epoch": 0.164, + "grad_norm": 0.09422667324542999, + "learning_rate": 0.00034921514788478915, + "loss": 3.6129, + "step": 16400 + }, + { + "epoch": 0.165, + "grad_norm": 0.1105048656463623, + "learning_rate": 0.0003481553119113957, + "loss": 3.6126, + "step": 16500 + }, + { + "epoch": 0.166, + "grad_norm": 0.11182115972042084, + "learning_rate": 0.00034710506725031166, + "loss": 3.6077, + "step": 16600 + }, + { + "epoch": 0.167, + "grad_norm": 0.15074919164180756, + "learning_rate": 0.0003460642701029914, + "loss": 3.6093, + "step": 16700 + }, + { + "epoch": 0.168, + "grad_norm": 0.11445897072553635, + "learning_rate": 0.00034503277967117707, + "loss": 3.6077, + "step": 16800 + }, + { + "epoch": 0.169, + "grad_norm": 0.10784851759672165, + "learning_rate": 0.0003440104580768907, + "loss": 3.6088, + "step": 16900 + }, + { + "epoch": 0.17, + "grad_norm": 0.11717010289430618, + "learning_rate": 0.00034299717028501764, + "loss": 3.6059, + "step": 17000 + }, + { + "epoch": 0.17, + "eval_loss": 4.03125, + "eval_runtime": 33.0161, + "eval_samples_per_second": 18239.208, + "eval_steps_per_second": 17.84, + "step": 17000 + }, + { + "epoch": 0.171, + "grad_norm": 0.10456647723913193, + "learning_rate": 0.00034199278402838475, + "loss": 3.6086, + "step": 17100 + }, + { + "epoch": 0.172, + "grad_norm": 0.11512298136949539, + "learning_rate": 0.00034099716973523677, + "loss": 3.6081, + "step": 17200 + }, + { + "epoch": 0.173, + "grad_norm": 0.11558009684085846, + "learning_rate": 0.000340010200459023, + "loss": 3.605, + "step": 17300 + }, + { + "epoch": 0.174, + "grad_norm": 0.12961934506893158, + "learning_rate": 0.0003390317518104052, + "loss": 3.6042, + "step": 17400 + }, + { + "epoch": 0.175, + "grad_norm": 0.10916338860988617, + "learning_rate": 0.0003380617018914066, + "loss": 3.605, + "step": 17500 + }, + { + "epoch": 0.176, + "grad_norm": 0.09793013334274292, + "learning_rate": 0.00033709993123162105, + "loss": 3.6012, + "step": 17600 + }, + { + "epoch": 0.177, + "grad_norm": 0.09620773047208786, + "learning_rate": 0.0003361463227264072, + "loss": 3.6022, + "step": 17700 + }, + { + "epoch": 0.178, + "grad_norm": 0.1438409984111786, + "learning_rate": 0.0003352007615769955, + "loss": 3.5989, + "step": 17800 + }, + { + "epoch": 0.179, + "grad_norm": 0.09917107224464417, + "learning_rate": 0.0003342631352324378, + "loss": 3.5997, + "step": 17900 + }, + { + "epoch": 0.18, + "grad_norm": 0.1120196059346199, + "learning_rate": 0.0003333333333333333, + "loss": 3.5982, + "step": 18000 + }, + { + "epoch": 0.18, + "eval_loss": 4.02734375, + "eval_runtime": 32.9951, + "eval_samples_per_second": 18250.853, + "eval_steps_per_second": 17.851, + "step": 18000 + }, + { + "epoch": 0.181, + "grad_norm": 0.11737548559904099, + "learning_rate": 0.00033241124765726683, + "loss": 3.6, + "step": 18100 + }, + { + "epoch": 0.182, + "grad_norm": 0.09610620886087418, + "learning_rate": 0.00033149677206589795, + "loss": 3.5976, + "step": 18200 + }, + { + "epoch": 0.183, + "grad_norm": 0.0966159924864769, + "learning_rate": 0.00033058980245364314, + "loss": 3.5956, + "step": 18300 + }, + { + "epoch": 0.184, + "grad_norm": 0.12272274494171143, + "learning_rate": 0.00032969023669789354, + "loss": 3.5948, + "step": 18400 + }, + { + "epoch": 0.185, + "grad_norm": 0.12435866892337799, + "learning_rate": 0.0003287979746107146, + "loss": 3.5962, + "step": 18500 + }, + { + "epoch": 0.186, + "grad_norm": 0.13750095665454865, + "learning_rate": 0.0003279129178919765, + "loss": 3.5942, + "step": 18600 + }, + { + "epoch": 0.187, + "grad_norm": 0.09951788187026978, + "learning_rate": 0.00032703497008386434, + "loss": 3.5948, + "step": 18700 + }, + { + "epoch": 0.188, + "grad_norm": 0.17174014449119568, + "learning_rate": 0.0003261640365267211, + "loss": 3.5918, + "step": 18800 + }, + { + "epoch": 0.189, + "grad_norm": 0.09349057078361511, + "learning_rate": 0.0003253000243161777, + "loss": 3.5933, + "step": 18900 + }, + { + "epoch": 0.19, + "grad_norm": 0.13002406060695648, + "learning_rate": 0.0003244428422615251, + "loss": 3.5934, + "step": 19000 + }, + { + "epoch": 0.19, + "eval_loss": 4.01171875, + "eval_runtime": 33.1166, + "eval_samples_per_second": 18183.864, + "eval_steps_per_second": 17.786, + "step": 19000 + }, + { + "epoch": 0.191, + "grad_norm": 0.09189624339342117, + "learning_rate": 0.0003235924008452868, + "loss": 3.5931, + "step": 19100 + }, + { + "epoch": 0.192, + "grad_norm": 0.14783063530921936, + "learning_rate": 0.0003227486121839514, + "loss": 3.5908, + "step": 19200 + }, + { + "epoch": 0.193, + "grad_norm": 0.12302859127521515, + "learning_rate": 0.00032191138998982524, + "loss": 3.59, + "step": 19300 + }, + { + "epoch": 0.194, + "grad_norm": 0.11037351936101913, + "learning_rate": 0.0003210806495339678, + "loss": 3.5881, + "step": 19400 + }, + { + "epoch": 0.195, + "grad_norm": 0.10322829335927963, + "learning_rate": 0.00032025630761017425, + "loss": 3.5877, + "step": 19500 + }, + { + "epoch": 0.196, + "grad_norm": 0.09346406161785126, + "learning_rate": 0.00031943828249997, + "loss": 3.5847, + "step": 19600 + }, + { + "epoch": 0.197, + "grad_norm": 0.10106361657381058, + "learning_rate": 0.0003186264939385831, + "loss": 3.5857, + "step": 19700 + }, + { + "epoch": 0.198, + "grad_norm": 0.11678298562765121, + "learning_rate": 0.0003178208630818641, + "loss": 3.5852, + "step": 19800 + }, + { + "epoch": 0.199, + "grad_norm": 0.10398001223802567, + "learning_rate": 0.00031702131247412076, + "loss": 3.5837, + "step": 19900 + }, + { + "epoch": 0.2, + "grad_norm": 0.09851796180009842, + "learning_rate": 0.00031622776601683794, + "loss": 3.5844, + "step": 20000 + }, + { + "epoch": 0.2, + "eval_loss": 4.0078125, + "eval_runtime": 33.1319, + "eval_samples_per_second": 18175.498, + "eval_steps_per_second": 17.777, + "step": 20000 + }, + { + "epoch": 0.201, + "grad_norm": 0.11415958404541016, + "learning_rate": 0.00031544014893825583, + "loss": 3.5827, + "step": 20100 + }, + { + "epoch": 0.202, + "grad_norm": 0.14236243069171906, + "learning_rate": 0.0003146583877637763, + "loss": 3.5815, + "step": 20200 + }, + { + "epoch": 0.203, + "grad_norm": 0.10967651009559631, + "learning_rate": 0.0003138824102871722, + "loss": 3.5822, + "step": 20300 + }, + { + "epoch": 0.204, + "grad_norm": 0.0956553965806961, + "learning_rate": 0.0003131121455425748, + "loss": 3.5814, + "step": 20400 + }, + { + "epoch": 0.205, + "grad_norm": 0.10149464756250381, + "learning_rate": 0.0003123475237772121, + "loss": 3.5802, + "step": 20500 + }, + { + "epoch": 0.206, + "grad_norm": 0.12360329180955887, + "learning_rate": 0.0003115884764248779, + "loss": 3.5806, + "step": 20600 + }, + { + "epoch": 0.207, + "grad_norm": 0.09695049375295639, + "learning_rate": 0.0003108349360801046, + "loss": 3.5792, + "step": 20700 + }, + { + "epoch": 0.208, + "grad_norm": 0.09988222271203995, + "learning_rate": 0.0003100868364730211, + "loss": 3.5786, + "step": 20800 + }, + { + "epoch": 0.209, + "grad_norm": 0.10975050926208496, + "learning_rate": 0.000309344112444873, + "loss": 3.5775, + "step": 20900 + }, + { + "epoch": 0.21, + "grad_norm": 0.11329706013202667, + "learning_rate": 0.00030860669992418383, + "loss": 3.579, + "step": 21000 + }, + { + "epoch": 0.21, + "eval_loss": 3.99609375, + "eval_runtime": 32.9688, + "eval_samples_per_second": 18265.41, + "eval_steps_per_second": 17.865, + "step": 21000 + }, + { + "epoch": 0.211, + "grad_norm": 0.10806494951248169, + "learning_rate": 0.00030787453590353956, + "loss": 3.5766, + "step": 21100 + }, + { + "epoch": 0.212, + "grad_norm": 0.10181587189435959, + "learning_rate": 0.0003071475584169756, + "loss": 3.578, + "step": 21200 + }, + { + "epoch": 0.213, + "grad_norm": 0.1141391471028328, + "learning_rate": 0.00030642570651794775, + "loss": 3.5747, + "step": 21300 + }, + { + "epoch": 0.214, + "grad_norm": 0.0977480486035347, + "learning_rate": 0.00030570892025787156, + "loss": 3.5767, + "step": 21400 + }, + { + "epoch": 0.215, + "grad_norm": 0.12418569624423981, + "learning_rate": 0.00030499714066520935, + "loss": 3.5739, + "step": 21500 + }, + { + "epoch": 0.216, + "grad_norm": 0.10716783255338669, + "learning_rate": 0.00030429030972509223, + "loss": 3.5754, + "step": 21600 + }, + { + "epoch": 0.217, + "grad_norm": 0.11646725982427597, + "learning_rate": 0.0003035883703594582, + "loss": 3.5718, + "step": 21700 + }, + { + "epoch": 0.218, + "grad_norm": 0.10092218220233917, + "learning_rate": 0.00030289126640769133, + "loss": 3.575, + "step": 21800 + }, + { + "epoch": 0.219, + "grad_norm": 0.10740833729505539, + "learning_rate": 0.0003021989426077497, + "loss": 3.5721, + "step": 21900 + }, + { + "epoch": 0.22, + "grad_norm": 0.11317898333072662, + "learning_rate": 0.00030151134457776364, + "loss": 3.5709, + "step": 22000 + }, + { + "epoch": 0.22, + "eval_loss": 3.9921875, + "eval_runtime": 32.9711, + "eval_samples_per_second": 18264.112, + "eval_steps_per_second": 17.864, + "step": 22000 + }, + { + "epoch": 0.221, + "grad_norm": 0.10360657423734665, + "learning_rate": 0.0003008284187980934, + "loss": 3.57, + "step": 22100 + }, + { + "epoch": 0.222, + "grad_norm": 0.12486978620290756, + "learning_rate": 0.0003001501125938321, + "loss": 3.5688, + "step": 22200 + }, + { + "epoch": 0.223, + "grad_norm": 0.09880336374044418, + "learning_rate": 0.00029947637411773995, + "loss": 3.5679, + "step": 22300 + }, + { + "epoch": 0.224, + "grad_norm": 0.116917684674263, + "learning_rate": 0.00029880715233359837, + "loss": 3.5674, + "step": 22400 + }, + { + "epoch": 0.225, + "grad_norm": 0.10839434713125229, + "learning_rate": 0.00029814239699997195, + "loss": 3.5662, + "step": 22500 + }, + { + "epoch": 0.226, + "grad_norm": 0.12070871144533157, + "learning_rate": 0.0002974820586543648, + "loss": 3.5661, + "step": 22600 + }, + { + "epoch": 0.227, + "grad_norm": 0.1208093985915184, + "learning_rate": 0.0002968260885977624, + "loss": 3.5677, + "step": 22700 + }, + { + "epoch": 0.228, + "grad_norm": 0.1058797761797905, + "learning_rate": 0.00029617443887954616, + "loss": 3.5673, + "step": 22800 + }, + { + "epoch": 0.229, + "grad_norm": 0.10531262308359146, + "learning_rate": 0.00029552706228277086, + "loss": 3.5658, + "step": 22900 + }, + { + "epoch": 0.23, + "grad_norm": 0.09782661497592926, + "learning_rate": 0.0002948839123097943, + "loss": 3.5664, + "step": 23000 + }, + { + "epoch": 0.23, + "eval_loss": 3.98828125, + "eval_runtime": 32.9558, + "eval_samples_per_second": 18272.61, + "eval_steps_per_second": 17.872, + "step": 23000 + }, + { + "epoch": 0.231, + "grad_norm": 0.09744828194379807, + "learning_rate": 0.00029424494316824986, + "loss": 3.5646, + "step": 23100 + }, + { + "epoch": 0.232, + "grad_norm": 0.10593089461326599, + "learning_rate": 0.00029361010975735173, + "loss": 3.5659, + "step": 23200 + }, + { + "epoch": 0.233, + "grad_norm": 0.13277162611484528, + "learning_rate": 0.0002929793676545238, + "loss": 3.563, + "step": 23300 + }, + { + "epoch": 0.234, + "grad_norm": 0.09761735796928406, + "learning_rate": 0.0002923526731023431, + "loss": 3.5636, + "step": 23400 + }, + { + "epoch": 0.235, + "grad_norm": 0.10143497586250305, + "learning_rate": 0.0002917299829957891, + "loss": 3.5624, + "step": 23500 + }, + { + "epoch": 0.236, + "grad_norm": 0.09824883192777634, + "learning_rate": 0.000291111254869791, + "loss": 3.5625, + "step": 23600 + }, + { + "epoch": 0.237, + "grad_norm": 0.1004340797662735, + "learning_rate": 0.0002904964468870634, + "loss": 3.5599, + "step": 23700 + }, + { + "epoch": 0.238, + "grad_norm": 0.10824216157197952, + "learning_rate": 0.00028988551782622426, + "loss": 3.5618, + "step": 23800 + }, + { + "epoch": 0.239, + "grad_norm": 0.09485534578561783, + "learning_rate": 0.0002892784270701859, + "loss": 3.563, + "step": 23900 + }, + { + "epoch": 0.24, + "grad_norm": 0.09868928045034409, + "learning_rate": 0.00028867513459481295, + "loss": 3.5597, + "step": 24000 + }, + { + "epoch": 0.24, + "eval_loss": 3.984375, + "eval_runtime": 33.116, + "eval_samples_per_second": 18184.205, + "eval_steps_per_second": 17.786, + "step": 24000 + }, + { + "epoch": 0.241, + "grad_norm": 0.10118967294692993, + "learning_rate": 0.00028807560095783866, + "loss": 3.5607, + "step": 24100 + }, + { + "epoch": 0.242, + "grad_norm": 0.11375439912080765, + "learning_rate": 0.00028747978728803456, + "loss": 3.5614, + "step": 24200 + }, + { + "epoch": 0.243, + "grad_norm": 0.1017841324210167, + "learning_rate": 0.00028688765527462345, + "loss": 3.5578, + "step": 24300 + }, + { + "epoch": 0.244, + "grad_norm": 0.10626488924026489, + "learning_rate": 0.0002862991671569341, + "loss": 3.5591, + "step": 24400 + }, + { + "epoch": 0.245, + "grad_norm": 0.12246066331863403, + "learning_rate": 0.0002857142857142857, + "loss": 3.5587, + "step": 24500 + }, + { + "epoch": 0.246, + "grad_norm": 0.11167100071907043, + "learning_rate": 0.0002851329742561005, + "loss": 3.5573, + "step": 24600 + }, + { + "epoch": 0.247, + "grad_norm": 0.12268108874559402, + "learning_rate": 0.0002845551966122361, + "loss": 3.5569, + "step": 24700 + }, + { + "epoch": 0.248, + "grad_norm": 0.09834783524274826, + "learning_rate": 0.0002839809171235324, + "loss": 3.5551, + "step": 24800 + }, + { + "epoch": 0.249, + "grad_norm": 0.11501438915729523, + "learning_rate": 0.00028341010063256787, + "loss": 3.5549, + "step": 24900 + }, + { + "epoch": 0.25, + "grad_norm": 0.12867997586727142, + "learning_rate": 0.000282842712474619, + "loss": 3.5538, + "step": 25000 + }, + { + "epoch": 0.25, + "eval_loss": 3.978515625, + "eval_runtime": 33.0607, + "eval_samples_per_second": 18214.619, + "eval_steps_per_second": 17.816, + "step": 25000 + }, + { + "epoch": 0.251, + "grad_norm": 0.09644894301891327, + "learning_rate": 0.0002822787184688183, + "loss": 3.556, + "step": 25100 + }, + { + "epoch": 0.252, + "grad_norm": 0.10058575868606567, + "learning_rate": 0.0002817180849095055, + "loss": 3.5537, + "step": 25200 + }, + { + "epoch": 0.253, + "grad_norm": 0.10156785696744919, + "learning_rate": 0.0002811607785577666, + "loss": 3.5531, + "step": 25300 + }, + { + "epoch": 0.254, + "grad_norm": 0.0953749492764473, + "learning_rate": 0.0002806067666331569, + "loss": 3.5514, + "step": 25400 + }, + { + "epoch": 0.255, + "grad_norm": 0.1226545125246048, + "learning_rate": 0.00028005601680560193, + "loss": 3.5512, + "step": 25500 + }, + { + "epoch": 0.256, + "grad_norm": 0.09581299126148224, + "learning_rate": 0.00027950849718747374, + "loss": 3.5515, + "step": 25600 + }, + { + "epoch": 0.257, + "grad_norm": 0.12539328634738922, + "learning_rate": 0.0002789641763258353, + "loss": 3.5513, + "step": 25700 + }, + { + "epoch": 0.258, + "grad_norm": 0.12774671614170074, + "learning_rate": 0.0002784230231948523, + "loss": 3.552, + "step": 25800 + }, + { + "epoch": 0.259, + "grad_norm": 0.10204903036355972, + "learning_rate": 0.00027788500718836423, + "loss": 3.5499, + "step": 25900 + }, + { + "epoch": 0.26, + "grad_norm": 0.09892131388187408, + "learning_rate": 0.0002773500981126146, + "loss": 3.551, + "step": 26000 + }, + { + "epoch": 0.26, + "eval_loss": 3.98046875, + "eval_runtime": 32.9829, + "eval_samples_per_second": 18257.575, + "eval_steps_per_second": 17.858, + "step": 26000 + }, + { + "epoch": 0.261, + "grad_norm": 0.09041019529104233, + "learning_rate": 0.0002768182661791332, + "loss": 3.5483, + "step": 26100 + }, + { + "epoch": 0.262, + "grad_norm": 0.09553086757659912, + "learning_rate": 0.00027628948199776883, + "loss": 3.546, + "step": 26200 + }, + { + "epoch": 0.263, + "grad_norm": 0.09868323802947998, + "learning_rate": 0.00027576371656986686, + "loss": 3.5477, + "step": 26300 + }, + { + "epoch": 0.264, + "grad_norm": 0.12876607477664948, + "learning_rate": 0.00027524094128159016, + "loss": 3.5483, + "step": 26400 + }, + { + "epoch": 0.265, + "grad_norm": 0.09975658357143402, + "learning_rate": 0.0002747211278973781, + "loss": 3.5484, + "step": 26500 + }, + { + "epoch": 0.266, + "grad_norm": 0.12821029126644135, + "learning_rate": 0.0002742042485535409, + "loss": 3.5471, + "step": 26600 + }, + { + "epoch": 0.267, + "grad_norm": 0.0922236442565918, + "learning_rate": 0.00027369027575198666, + "loss": 3.5473, + "step": 26700 + }, + { + "epoch": 0.268, + "grad_norm": 0.11472651362419128, + "learning_rate": 0.0002731791823540765, + "loss": 3.5465, + "step": 26800 + }, + { + "epoch": 0.269, + "grad_norm": 0.10681800544261932, + "learning_rate": 0.0002726709415746059, + "loss": 3.544, + "step": 26900 + }, + { + "epoch": 0.27, + "grad_norm": 0.1048787459731102, + "learning_rate": 0.0002721655269759087, + "loss": 3.545, + "step": 27000 + }, + { + "epoch": 0.27, + "eval_loss": 3.97265625, + "eval_runtime": 32.9837, + "eval_samples_per_second": 18257.114, + "eval_steps_per_second": 17.857, + "step": 27000 + }, + { + "epoch": 0.271, + "grad_norm": 0.10347763448953629, + "learning_rate": 0.00027166291246208065, + "loss": 3.5436, + "step": 27100 + }, + { + "epoch": 0.272, + "grad_norm": 0.10141801089048386, + "learning_rate": 0.0002711630722733202, + "loss": 3.5438, + "step": 27200 + }, + { + "epoch": 0.273, + "grad_norm": 0.09036429226398468, + "learning_rate": 0.00027066598098038336, + "loss": 3.543, + "step": 27300 + }, + { + "epoch": 0.274, + "grad_norm": 0.09406576305627823, + "learning_rate": 0.0002701716134791496, + "loss": 3.5427, + "step": 27400 + }, + { + "epoch": 0.275, + "grad_norm": 0.09352946281433105, + "learning_rate": 0.00026967994498529687, + "loss": 3.5417, + "step": 27500 + }, + { + "epoch": 0.276, + "grad_norm": 0.09936648607254028, + "learning_rate": 0.00026919095102908273, + "loss": 3.5428, + "step": 27600 + }, + { + "epoch": 0.277, + "grad_norm": 0.10206812620162964, + "learning_rate": 0.00026870460745022953, + "loss": 3.5424, + "step": 27700 + }, + { + "epoch": 0.278, + "grad_norm": 0.11528582870960236, + "learning_rate": 0.00026822089039291, + "loss": 3.5432, + "step": 27800 + }, + { + "epoch": 0.279, + "grad_norm": 0.09294348955154419, + "learning_rate": 0.00026773977630083294, + "loss": 3.5409, + "step": 27900 + }, + { + "epoch": 0.28, + "grad_norm": 0.10319820046424866, + "learning_rate": 0.0002672612419124244, + "loss": 3.5414, + "step": 28000 + }, + { + "epoch": 0.28, + "eval_loss": 3.962890625, + "eval_runtime": 32.9656, + "eval_samples_per_second": 18267.183, + "eval_steps_per_second": 17.867, + "step": 28000 + }, + { + "epoch": 0.281, + "grad_norm": 0.12130926549434662, + "learning_rate": 0.0002667852642561041, + "loss": 3.5419, + "step": 28100 + }, + { + "epoch": 0.282, + "grad_norm": 0.1062818095088005, + "learning_rate": 0.00026631182064565373, + "loss": 3.5414, + "step": 28200 + }, + { + "epoch": 0.283, + "grad_norm": 0.10452710092067719, + "learning_rate": 0.0002658408886756753, + "loss": 3.5429, + "step": 28300 + }, + { + "epoch": 0.284, + "grad_norm": 0.15059900283813477, + "learning_rate": 0.0002653724462171376, + "loss": 3.5387, + "step": 28400 + }, + { + "epoch": 0.285, + "grad_norm": 0.1282389611005783, + "learning_rate": 0.00026490647141300875, + "loss": 3.5371, + "step": 28500 + }, + { + "epoch": 0.286, + "grad_norm": 0.09370238333940506, + "learning_rate": 0.0002644429426739725, + "loss": 3.5379, + "step": 28600 + }, + { + "epoch": 0.287, + "grad_norm": 0.11192005127668381, + "learning_rate": 0.00026398183867422733, + "loss": 3.5394, + "step": 28700 + }, + { + "epoch": 0.288, + "grad_norm": 0.11204060167074203, + "learning_rate": 0.00026352313834736497, + "loss": 3.5378, + "step": 28800 + }, + { + "epoch": 0.289, + "grad_norm": 0.11505492776632309, + "learning_rate": 0.0002630668208823282, + "loss": 3.5368, + "step": 28900 + }, + { + "epoch": 0.29, + "grad_norm": 0.09516780078411102, + "learning_rate": 0.0002626128657194451, + "loss": 3.5368, + "step": 29000 + }, + { + "epoch": 0.29, + "eval_loss": 3.962890625, + "eval_runtime": 32.9727, + "eval_samples_per_second": 18263.215, + "eval_steps_per_second": 17.863, + "step": 29000 + }, + { + "epoch": 0.291, + "grad_norm": 0.1098528504371643, + "learning_rate": 0.00026216125254653817, + "loss": 3.5356, + "step": 29100 + }, + { + "epoch": 0.292, + "grad_norm": 0.09026819467544556, + "learning_rate": 0.0002617119612951068, + "loss": 3.5342, + "step": 29200 + }, + { + "epoch": 0.293, + "grad_norm": 0.09625507891178131, + "learning_rate": 0.00026126497213658205, + "loss": 3.538, + "step": 29300 + }, + { + "epoch": 0.294, + "grad_norm": 0.13931472599506378, + "learning_rate": 0.00026082026547865055, + "loss": 3.5355, + "step": 29400 + }, + { + "epoch": 0.295, + "grad_norm": 0.09799457341432571, + "learning_rate": 0.0002603778219616478, + "loss": 3.5323, + "step": 29500 + }, + { + "epoch": 0.296, + "grad_norm": 0.10035666823387146, + "learning_rate": 0.00025993762245501815, + "loss": 3.5353, + "step": 29600 + }, + { + "epoch": 0.297, + "grad_norm": 0.09533281624317169, + "learning_rate": 0.000259499648053841, + "loss": 3.5361, + "step": 29700 + }, + { + "epoch": 0.298, + "grad_norm": 0.1048363596200943, + "learning_rate": 0.00025906388007541984, + "loss": 3.533, + "step": 29800 + }, + { + "epoch": 0.299, + "grad_norm": 0.1037365198135376, + "learning_rate": 0.00025863030005593586, + "loss": 3.5328, + "step": 29900 + }, + { + "epoch": 0.3, + "grad_norm": 0.09484585374593735, + "learning_rate": 0.0002581988897471611, + "loss": 3.5336, + "step": 30000 + }, + { + "epoch": 0.3, + "eval_loss": 3.958984375, + "eval_runtime": 32.9766, + "eval_samples_per_second": 18261.09, + "eval_steps_per_second": 17.861, + "step": 30000 + }, + { + "epoch": 0.301, + "grad_norm": 0.09425152838230133, + "learning_rate": 0.00025776963111323354, + "loss": 3.5315, + "step": 30100 + }, + { + "epoch": 0.302, + "grad_norm": 0.09814691543579102, + "learning_rate": 0.0002573425063274894, + "loss": 3.5347, + "step": 30200 + }, + { + "epoch": 0.303, + "grad_norm": 0.1039854884147644, + "learning_rate": 0.00025691749776935395, + "loss": 3.5307, + "step": 30300 + }, + { + "epoch": 0.304, + "grad_norm": 0.11937417089939117, + "learning_rate": 0.0002564945880212886, + "loss": 3.5315, + "step": 30400 + }, + { + "epoch": 0.305, + "grad_norm": 0.10565177351236343, + "learning_rate": 0.000256073759865792, + "loss": 3.5316, + "step": 30500 + }, + { + "epoch": 0.306, + "grad_norm": 0.12661132216453552, + "learning_rate": 0.00025565499628245683, + "loss": 3.5289, + "step": 30600 + }, + { + "epoch": 0.307, + "grad_norm": 0.10317707061767578, + "learning_rate": 0.00025523828044507794, + "loss": 3.5307, + "step": 30700 + }, + { + "epoch": 0.308, + "grad_norm": 0.10436379164457321, + "learning_rate": 0.00025482359571881276, + "loss": 3.5295, + "step": 30800 + }, + { + "epoch": 0.309, + "grad_norm": 0.09864766150712967, + "learning_rate": 0.0002544109256573922, + "loss": 3.53, + "step": 30900 + }, + { + "epoch": 0.31, + "grad_norm": 0.0929613932967186, + "learning_rate": 0.000254000254000381, + "loss": 3.5272, + "step": 31000 + }, + { + "epoch": 0.31, + "eval_loss": 3.955078125, + "eval_runtime": 33.0115, + "eval_samples_per_second": 18241.739, + "eval_steps_per_second": 17.842, + "step": 31000 + }, + { + "epoch": 0.311, + "grad_norm": 0.09546711295843124, + "learning_rate": 0.00025359156467048686, + "loss": 3.528, + "step": 31100 + }, + { + "epoch": 0.312, + "grad_norm": 0.09321217238903046, + "learning_rate": 0.00025318484177091664, + "loss": 3.5296, + "step": 31200 + }, + { + "epoch": 0.313, + "grad_norm": 0.11899935454130173, + "learning_rate": 0.00025278006958277935, + "loss": 3.528, + "step": 31300 + }, + { + "epoch": 0.314, + "grad_norm": 0.09187714010477066, + "learning_rate": 0.00025237723256253436, + "loss": 3.5265, + "step": 31400 + }, + { + "epoch": 0.315, + "grad_norm": 0.09416347742080688, + "learning_rate": 0.0002519763153394848, + "loss": 3.5277, + "step": 31500 + }, + { + "epoch": 0.316, + "grad_norm": 0.09648656100034714, + "learning_rate": 0.0002515773027133138, + "loss": 3.5246, + "step": 31600 + }, + { + "epoch": 0.317, + "grad_norm": 0.09706927835941315, + "learning_rate": 0.0002511801796516642, + "loss": 3.5266, + "step": 31700 + }, + { + "epoch": 0.318, + "grad_norm": 0.10058547556400299, + "learning_rate": 0.00025078493128775957, + "loss": 3.5249, + "step": 31800 + }, + { + "epoch": 0.319, + "grad_norm": 0.11078114062547684, + "learning_rate": 0.0002503915429180672, + "loss": 3.5258, + "step": 31900 + }, + { + "epoch": 0.32, + "grad_norm": 0.09327193349599838, + "learning_rate": 0.00025, + "loss": 3.5257, + "step": 32000 + }, + { + "epoch": 0.32, + "eval_loss": 3.953125, + "eval_runtime": 32.9821, + "eval_samples_per_second": 18258.033, + "eval_steps_per_second": 17.858, + "step": 32000 + }, + { + "epoch": 0.321, + "grad_norm": 0.10520542412996292, + "learning_rate": 0.0002496102881496589, + "loss": 3.5264, + "step": 32100 + }, + { + "epoch": 0.322, + "grad_norm": 0.11748110502958298, + "learning_rate": 0.0002492223931396134, + "loss": 3.526, + "step": 32200 + }, + { + "epoch": 0.323, + "grad_norm": 0.11243593692779541, + "learning_rate": 0.0002488363008967198, + "loss": 3.5244, + "step": 32300 + }, + { + "epoch": 0.324, + "grad_norm": 0.09914110600948334, + "learning_rate": 0.00024845199749997667, + "loss": 3.5249, + "step": 32400 + }, + { + "epoch": 0.325, + "grad_norm": 0.0963945984840393, + "learning_rate": 0.00024806946917841694, + "loss": 3.5244, + "step": 32500 + }, + { + "epoch": 0.326, + "grad_norm": 0.09173919260501862, + "learning_rate": 0.00024768870230903496, + "loss": 3.5236, + "step": 32600 + }, + { + "epoch": 0.327, + "grad_norm": 0.1360139101743698, + "learning_rate": 0.00024730968341474897, + "loss": 3.5213, + "step": 32700 + }, + { + "epoch": 0.328, + "grad_norm": 0.09488269686698914, + "learning_rate": 0.0002469323991623974, + "loss": 3.5235, + "step": 32800 + }, + { + "epoch": 0.329, + "grad_norm": 0.1184873878955841, + "learning_rate": 0.00024655683636076893, + "loss": 3.5233, + "step": 32900 + }, + { + "epoch": 0.33, + "grad_norm": 0.09744580835103989, + "learning_rate": 0.00024618298195866543, + "loss": 3.5208, + "step": 33000 + }, + { + "epoch": 0.33, + "eval_loss": 3.951171875, + "eval_runtime": 32.9597, + "eval_samples_per_second": 18270.423, + "eval_steps_per_second": 17.87, + "step": 33000 + }, + { + "epoch": 0.331, + "grad_norm": 0.09271999448537827, + "learning_rate": 0.0002458108230429969, + "loss": 3.5197, + "step": 33100 + }, + { + "epoch": 0.332, + "grad_norm": 0.12717638909816742, + "learning_rate": 0.00024544034683690797, + "loss": 3.5194, + "step": 33200 + }, + { + "epoch": 0.333, + "grad_norm": 0.10852134227752686, + "learning_rate": 0.0002450715406979359, + "loss": 3.5196, + "step": 33300 + }, + { + "epoch": 0.334, + "grad_norm": 0.10992783308029175, + "learning_rate": 0.0002447043921161982, + "loss": 3.5216, + "step": 33400 + }, + { + "epoch": 0.335, + "grad_norm": 0.1337759792804718, + "learning_rate": 0.0002443388887126105, + "loss": 3.5205, + "step": 33500 + }, + { + "epoch": 0.336, + "grad_norm": 0.11975846439599991, + "learning_rate": 0.00024397501823713327, + "loss": 3.5191, + "step": 33600 + }, + { + "epoch": 0.337, + "grad_norm": 0.14458173513412476, + "learning_rate": 0.00024361276856704796, + "loss": 3.5203, + "step": 33700 + }, + { + "epoch": 0.338, + "grad_norm": 0.09570927172899246, + "learning_rate": 0.00024325212770525995, + "loss": 3.5184, + "step": 33800 + }, + { + "epoch": 0.339, + "grad_norm": 0.11703662574291229, + "learning_rate": 0.00024289308377863016, + "loss": 3.5176, + "step": 33900 + }, + { + "epoch": 0.34, + "grad_norm": 0.09540607035160065, + "learning_rate": 0.00024253562503633296, + "loss": 3.5168, + "step": 34000 + }, + { + "epoch": 0.34, + "eval_loss": 3.943359375, + "eval_runtime": 32.983, + "eval_samples_per_second": 18257.51, + "eval_steps_per_second": 17.858, + "step": 34000 + }, + { + "epoch": 0.341, + "grad_norm": 0.09772911667823792, + "learning_rate": 0.00024217973984824137, + "loss": 3.5173, + "step": 34100 + }, + { + "epoch": 0.342, + "grad_norm": 0.10085777193307877, + "learning_rate": 0.0002418254167033372, + "loss": 3.5179, + "step": 34200 + }, + { + "epoch": 0.343, + "grad_norm": 0.10360076278448105, + "learning_rate": 0.0002414726442081476, + "loss": 3.5171, + "step": 34300 + }, + { + "epoch": 0.344, + "grad_norm": 0.10194404423236847, + "learning_rate": 0.0002411214110852061, + "loss": 3.5145, + "step": 34400 + }, + { + "epoch": 0.345, + "grad_norm": 0.1001388356089592, + "learning_rate": 0.0002407717061715384, + "loss": 3.5177, + "step": 34500 + }, + { + "epoch": 0.346, + "grad_norm": 0.10188458114862442, + "learning_rate": 0.00024042351841717252, + "loss": 3.5156, + "step": 34600 + }, + { + "epoch": 0.347, + "grad_norm": 0.10750128328800201, + "learning_rate": 0.00024007683688367183, + "loss": 3.5192, + "step": 34700 + }, + { + "epoch": 0.348, + "grad_norm": 0.12118516862392426, + "learning_rate": 0.0002397316507426921, + "loss": 3.517, + "step": 34800 + }, + { + "epoch": 0.349, + "grad_norm": 0.09460828453302383, + "learning_rate": 0.0002393879492745607, + "loss": 3.5176, + "step": 34900 + }, + { + "epoch": 0.35, + "grad_norm": 0.10879145562648773, + "learning_rate": 0.00023904572186687873, + "loss": 3.5135, + "step": 35000 + }, + { + "epoch": 0.35, + "eval_loss": 3.94140625, + "eval_runtime": 32.9909, + "eval_samples_per_second": 18253.178, + "eval_steps_per_second": 17.853, + "step": 35000 + }, + { + "epoch": 0.351, + "grad_norm": 0.10676248371601105, + "learning_rate": 0.00023870495801314433, + "loss": 3.5156, + "step": 35100 + }, + { + "epoch": 0.352, + "grad_norm": 0.09372402727603912, + "learning_rate": 0.00023836564731139807, + "loss": 3.5155, + "step": 35200 + }, + { + "epoch": 0.353, + "grad_norm": 0.09432759135961533, + "learning_rate": 0.00023802777946288955, + "loss": 3.5148, + "step": 35300 + }, + { + "epoch": 0.354, + "grad_norm": 0.13439424335956573, + "learning_rate": 0.00023769134427076416, + "loss": 3.5149, + "step": 35400 + }, + { + "epoch": 0.355, + "grad_norm": 0.09214833378791809, + "learning_rate": 0.00023735633163877065, + "loss": 3.5141, + "step": 35500 + }, + { + "epoch": 0.356, + "grad_norm": 0.08910084515810013, + "learning_rate": 0.00023702273156998862, + "loss": 3.513, + "step": 35600 + }, + { + "epoch": 0.357, + "grad_norm": 0.09308894723653793, + "learning_rate": 0.00023669053416557542, + "loss": 3.5136, + "step": 35700 + }, + { + "epoch": 0.358, + "grad_norm": 0.09210114926099777, + "learning_rate": 0.00023635972962353274, + "loss": 3.513, + "step": 35800 + }, + { + "epoch": 0.359, + "grad_norm": 0.0928545594215393, + "learning_rate": 0.0002360303082374915, + "loss": 3.5111, + "step": 35900 + }, + { + "epoch": 0.36, + "grad_norm": 0.09326168894767761, + "learning_rate": 0.00023570226039551587, + "loss": 3.5124, + "step": 36000 + }, + { + "epoch": 0.36, + "eval_loss": 3.9453125, + "eval_runtime": 32.9906, + "eval_samples_per_second": 18253.307, + "eval_steps_per_second": 17.854, + "step": 36000 + }, + { + "epoch": 0.361, + "grad_norm": 0.11911537498235703, + "learning_rate": 0.00023537557657892522, + "loss": 3.5123, + "step": 36100 + }, + { + "epoch": 0.362, + "grad_norm": 0.12963728606700897, + "learning_rate": 0.00023505024736113423, + "loss": 3.5114, + "step": 36200 + }, + { + "epoch": 0.363, + "grad_norm": 0.10376954078674316, + "learning_rate": 0.00023472626340651013, + "loss": 3.5101, + "step": 36300 + }, + { + "epoch": 0.364, + "grad_norm": 0.09793874621391296, + "learning_rate": 0.00023440361546924772, + "loss": 3.5112, + "step": 36400 + }, + { + "epoch": 0.365, + "grad_norm": 0.1108843982219696, + "learning_rate": 0.00023408229439226115, + "loss": 3.509, + "step": 36500 + }, + { + "epoch": 0.366, + "grad_norm": 0.17180891335010529, + "learning_rate": 0.0002337622911060922, + "loss": 3.5084, + "step": 36600 + }, + { + "epoch": 0.367, + "grad_norm": 0.12893500924110413, + "learning_rate": 0.00023344359662783541, + "loss": 3.5105, + "step": 36700 + }, + { + "epoch": 0.368, + "grad_norm": 0.09860619902610779, + "learning_rate": 0.00023312620206007844, + "loss": 3.5104, + "step": 36800 + }, + { + "epoch": 0.369, + "grad_norm": 0.12415798008441925, + "learning_rate": 0.00023281009858985938, + "loss": 3.511, + "step": 36900 + }, + { + "epoch": 0.37, + "grad_norm": 0.08730453252792358, + "learning_rate": 0.00023249527748763857, + "loss": 3.51, + "step": 37000 + }, + { + "epoch": 0.37, + "eval_loss": 3.94140625, + "eval_runtime": 32.9875, + "eval_samples_per_second": 18255.033, + "eval_steps_per_second": 17.855, + "step": 37000 + }, + { + "epoch": 0.371, + "grad_norm": 0.09179120510816574, + "learning_rate": 0.000232181730106286, + "loss": 3.5092, + "step": 37100 + }, + { + "epoch": 0.372, + "grad_norm": 0.09930966049432755, + "learning_rate": 0.00023186944788008412, + "loss": 3.5109, + "step": 37200 + }, + { + "epoch": 0.373, + "grad_norm": 0.09958541393280029, + "learning_rate": 0.00023155842232374464, + "loss": 3.5109, + "step": 37300 + }, + { + "epoch": 0.374, + "grad_norm": 0.12271804362535477, + "learning_rate": 0.00023124864503144013, + "loss": 3.5105, + "step": 37400 + }, + { + "epoch": 0.375, + "grad_norm": 0.09509623795747757, + "learning_rate": 0.00023094010767585029, + "loss": 3.5071, + "step": 37500 + }, + { + "epoch": 0.376, + "grad_norm": 0.10307575762271881, + "learning_rate": 0.00023063280200722128, + "loss": 3.509, + "step": 37600 + }, + { + "epoch": 0.377, + "grad_norm": 0.11047662794589996, + "learning_rate": 0.00023032671985243937, + "loss": 3.507, + "step": 37700 + }, + { + "epoch": 0.378, + "grad_norm": 0.1369996964931488, + "learning_rate": 0.00023002185311411807, + "loss": 3.5077, + "step": 37800 + }, + { + "epoch": 0.379, + "grad_norm": 0.09368377178907394, + "learning_rate": 0.0002297181937696983, + "loss": 3.5079, + "step": 37900 + }, + { + "epoch": 0.38, + "grad_norm": 0.0933164581656456, + "learning_rate": 0.00022941573387056174, + "loss": 3.5061, + "step": 38000 + }, + { + "epoch": 0.38, + "eval_loss": 3.94140625, + "eval_runtime": 33.0024, + "eval_samples_per_second": 18246.79, + "eval_steps_per_second": 17.847, + "step": 38000 + }, + { + "epoch": 0.381, + "grad_norm": 0.11005818843841553, + "learning_rate": 0.0002291144655411569, + "loss": 3.5067, + "step": 38100 + }, + { + "epoch": 0.382, + "grad_norm": 0.09861373901367188, + "learning_rate": 0.00022881438097813774, + "loss": 3.5034, + "step": 38200 + }, + { + "epoch": 0.383, + "grad_norm": 0.10625416785478592, + "learning_rate": 0.00022851547244951505, + "loss": 3.5034, + "step": 38300 + }, + { + "epoch": 0.384, + "grad_norm": 0.09815394133329391, + "learning_rate": 0.00022821773229381924, + "loss": 3.505, + "step": 38400 + }, + { + "epoch": 0.385, + "grad_norm": 0.11951077729463577, + "learning_rate": 0.0002279211529192759, + "loss": 3.5052, + "step": 38500 + }, + { + "epoch": 0.386, + "grad_norm": 0.09885293990373611, + "learning_rate": 0.0002276257268029927, + "loss": 3.5032, + "step": 38600 + }, + { + "epoch": 0.387, + "grad_norm": 0.10494934767484665, + "learning_rate": 0.0002273314464901578, + "loss": 3.5018, + "step": 38700 + }, + { + "epoch": 0.388, + "grad_norm": 0.1292721927165985, + "learning_rate": 0.0002270383045932499, + "loss": 3.5039, + "step": 38800 + }, + { + "epoch": 0.389, + "grad_norm": 0.0997849851846695, + "learning_rate": 0.00022674629379125912, + "loss": 3.5016, + "step": 38900 + }, + { + "epoch": 0.39, + "grad_norm": 0.1319294422864914, + "learning_rate": 0.00022645540682891913, + "loss": 3.5044, + "step": 39000 + }, + { + "epoch": 0.39, + "eval_loss": 3.931640625, + "eval_runtime": 32.9931, + "eval_samples_per_second": 18251.944, + "eval_steps_per_second": 17.852, + "step": 39000 + }, + { + "epoch": 0.391, + "grad_norm": 0.11860304325819016, + "learning_rate": 0.0002261656365159503, + "loss": 3.5038, + "step": 39100 + }, + { + "epoch": 0.392, + "grad_norm": 0.1081485003232956, + "learning_rate": 0.00022587697572631283, + "loss": 3.5016, + "step": 39200 + }, + { + "epoch": 0.393, + "grad_norm": 0.10177210718393326, + "learning_rate": 0.00022558941739747075, + "loss": 3.5034, + "step": 39300 + }, + { + "epoch": 0.394, + "grad_norm": 0.08930639922618866, + "learning_rate": 0.00022530295452966644, + "loss": 3.5026, + "step": 39400 + }, + { + "epoch": 0.395, + "grad_norm": 0.10102177411317825, + "learning_rate": 0.0002250175801852048, + "loss": 3.5022, + "step": 39500 + }, + { + "epoch": 0.396, + "grad_norm": 0.11118755489587784, + "learning_rate": 0.00022473328748774736, + "loss": 3.5019, + "step": 39600 + }, + { + "epoch": 0.397, + "grad_norm": 0.09080816060304642, + "learning_rate": 0.00022445006962161677, + "loss": 3.5018, + "step": 39700 + }, + { + "epoch": 0.398, + "grad_norm": 0.09106365591287613, + "learning_rate": 0.00022416791983111015, + "loss": 3.4995, + "step": 39800 + }, + { + "epoch": 0.399, + "grad_norm": 0.08732519298791885, + "learning_rate": 0.0002238868314198225, + "loss": 3.4991, + "step": 39900 + }, + { + "epoch": 0.4, + "grad_norm": 0.08918973803520203, + "learning_rate": 0.00022360679774997898, + "loss": 3.501, + "step": 40000 + }, + { + "epoch": 0.4, + "eval_loss": 3.931640625, + "eval_runtime": 33.0068, + "eval_samples_per_second": 18244.355, + "eval_steps_per_second": 17.845, + "step": 40000 + }, + { + "epoch": 0.401, + "grad_norm": 0.08663675934076309, + "learning_rate": 0.00022332781224177668, + "loss": 3.501, + "step": 40100 + }, + { + "epoch": 0.402, + "grad_norm": 0.09693952649831772, + "learning_rate": 0.00022304986837273525, + "loss": 3.5006, + "step": 40200 + }, + { + "epoch": 0.403, + "grad_norm": 0.11651097983121872, + "learning_rate": 0.00022277295967705653, + "loss": 3.4992, + "step": 40300 + }, + { + "epoch": 0.404, + "grad_norm": 0.08836641907691956, + "learning_rate": 0.0002224970797449924, + "loss": 3.4997, + "step": 40400 + }, + { + "epoch": 0.405, + "grad_norm": 0.10767354816198349, + "learning_rate": 0.0002222222222222222, + "loss": 3.4995, + "step": 40500 + }, + { + "epoch": 0.406, + "grad_norm": 0.09033215790987015, + "learning_rate": 0.00022194838080923766, + "loss": 3.4976, + "step": 40600 + }, + { + "epoch": 0.407, + "grad_norm": 0.0916823074221611, + "learning_rate": 0.00022167554926073632, + "loss": 3.4983, + "step": 40700 + }, + { + "epoch": 0.408, + "grad_norm": 0.13178223371505737, + "learning_rate": 0.00022140372138502385, + "loss": 3.4957, + "step": 40800 + }, + { + "epoch": 0.409, + "grad_norm": 0.1049867570400238, + "learning_rate": 0.00022113289104342323, + "loss": 3.4973, + "step": 40900 + }, + { + "epoch": 0.41, + "grad_norm": 0.09196602553129196, + "learning_rate": 0.00022086305214969308, + "loss": 3.4996, + "step": 41000 + }, + { + "epoch": 0.41, + "eval_loss": 3.9296875, + "eval_runtime": 32.9946, + "eval_samples_per_second": 18251.123, + "eval_steps_per_second": 17.851, + "step": 41000 + }, + { + "epoch": 0.411, + "grad_norm": 0.0933445394039154, + "learning_rate": 0.00022059419866945278, + "loss": 3.4972, + "step": 41100 + }, + { + "epoch": 0.412, + "grad_norm": 0.10621432214975357, + "learning_rate": 0.00022032632461961585, + "loss": 3.4988, + "step": 41200 + }, + { + "epoch": 0.413, + "grad_norm": 0.1379343420267105, + "learning_rate": 0.00022005942406783078, + "loss": 3.4983, + "step": 41300 + }, + { + "epoch": 0.414, + "grad_norm": 0.1172490119934082, + "learning_rate": 0.00021979349113192902, + "loss": 3.4954, + "step": 41400 + }, + { + "epoch": 0.415, + "grad_norm": 0.10637906193733215, + "learning_rate": 0.0002195285199793807, + "loss": 3.4974, + "step": 41500 + }, + { + "epoch": 0.416, + "grad_norm": 0.10279926657676697, + "learning_rate": 0.0002192645048267573, + "loss": 3.495, + "step": 41600 + }, + { + "epoch": 0.417, + "grad_norm": 0.13635499775409698, + "learning_rate": 0.00021900143993920144, + "loss": 3.4997, + "step": 41700 + }, + { + "epoch": 0.418, + "grad_norm": 0.11549071967601776, + "learning_rate": 0.00021873931962990357, + "loss": 3.4963, + "step": 41800 + }, + { + "epoch": 0.419, + "grad_norm": 0.11906278878450394, + "learning_rate": 0.00021847813825958586, + "loss": 3.4937, + "step": 41900 + }, + { + "epoch": 0.42, + "grad_norm": 0.08883855491876602, + "learning_rate": 0.0002182178902359924, + "loss": 3.4957, + "step": 42000 + }, + { + "epoch": 0.42, + "eval_loss": 3.9296875, + "eval_runtime": 32.9839, + "eval_samples_per_second": 18257.009, + "eval_steps_per_second": 17.857, + "step": 42000 + }, + { + "epoch": 0.421, + "grad_norm": 0.11999965459108353, + "learning_rate": 0.00021795857001338643, + "loss": 3.4951, + "step": 42100 + }, + { + "epoch": 0.422, + "grad_norm": 0.09602005034685135, + "learning_rate": 0.00021770017209205407, + "loss": 3.4948, + "step": 42200 + }, + { + "epoch": 0.423, + "grad_norm": 0.12526479363441467, + "learning_rate": 0.00021744269101781405, + "loss": 3.4963, + "step": 42300 + }, + { + "epoch": 0.424, + "grad_norm": 0.08893965929746628, + "learning_rate": 0.0002171861213815347, + "loss": 3.4941, + "step": 42400 + }, + { + "epoch": 0.425, + "grad_norm": 0.10459540784358978, + "learning_rate": 0.00021693045781865617, + "loss": 3.4968, + "step": 42500 + }, + { + "epoch": 0.426, + "grad_norm": 0.10754283517599106, + "learning_rate": 0.00021667569500871976, + "loss": 3.4934, + "step": 42600 + }, + { + "epoch": 0.427, + "grad_norm": 0.1145748421549797, + "learning_rate": 0.0002164218276749025, + "loss": 3.4972, + "step": 42700 + }, + { + "epoch": 0.428, + "grad_norm": 0.10786953568458557, + "learning_rate": 0.00021616885058355848, + "loss": 3.4929, + "step": 42800 + }, + { + "epoch": 0.429, + "grad_norm": 0.11820059269666672, + "learning_rate": 0.00021591675854376523, + "loss": 3.4935, + "step": 42900 + }, + { + "epoch": 0.43, + "grad_norm": 0.08727939426898956, + "learning_rate": 0.00021566554640687683, + "loss": 3.4932, + "step": 43000 + }, + { + "epoch": 0.43, + "eval_loss": 3.92578125, + "eval_runtime": 32.9903, + "eval_samples_per_second": 18253.464, + "eval_steps_per_second": 17.854, + "step": 43000 + }, + { + "epoch": 0.431, + "grad_norm": 0.09863859415054321, + "learning_rate": 0.00021541520906608185, + "loss": 3.4939, + "step": 43100 + }, + { + "epoch": 0.432, + "grad_norm": 0.10082012414932251, + "learning_rate": 0.0002151657414559676, + "loss": 3.4948, + "step": 43200 + }, + { + "epoch": 0.433, + "grad_norm": 0.15581797063350677, + "learning_rate": 0.00021491713855208944, + "loss": 3.4922, + "step": 43300 + }, + { + "epoch": 0.434, + "grad_norm": 0.08548469096422195, + "learning_rate": 0.00021466939537054594, + "loss": 3.4897, + "step": 43400 + }, + { + "epoch": 0.435, + "grad_norm": 0.09384721517562866, + "learning_rate": 0.00021442250696755897, + "loss": 3.4923, + "step": 43500 + }, + { + "epoch": 0.436, + "grad_norm": 0.11810631304979324, + "learning_rate": 0.00021417646843905965, + "loss": 3.4907, + "step": 43600 + }, + { + "epoch": 0.437, + "grad_norm": 0.10735498368740082, + "learning_rate": 0.00021393127492027903, + "loss": 3.4928, + "step": 43700 + }, + { + "epoch": 0.438, + "grad_norm": 0.09564588963985443, + "learning_rate": 0.00021368692158534413, + "loss": 3.491, + "step": 43800 + }, + { + "epoch": 0.439, + "grad_norm": 0.08804554492235184, + "learning_rate": 0.00021344340364687887, + "loss": 3.4908, + "step": 43900 + }, + { + "epoch": 0.44, + "grad_norm": 0.09846536815166473, + "learning_rate": 0.00021320071635561042, + "loss": 3.4934, + "step": 44000 + }, + { + "epoch": 0.44, + "eval_loss": 3.919921875, + "eval_runtime": 33.0158, + "eval_samples_per_second": 18239.397, + "eval_steps_per_second": 17.84, + "step": 44000 + }, + { + "epoch": 0.441, + "grad_norm": 0.09984950721263885, + "learning_rate": 0.00021295885499997996, + "loss": 3.4919, + "step": 44100 + }, + { + "epoch": 0.442, + "grad_norm": 0.09016887098550797, + "learning_rate": 0.00021271781490575853, + "loss": 3.4899, + "step": 44200 + }, + { + "epoch": 0.443, + "grad_norm": 0.08681657910346985, + "learning_rate": 0.00021247759143566762, + "loss": 3.4906, + "step": 44300 + }, + { + "epoch": 0.444, + "grad_norm": 0.09243223071098328, + "learning_rate": 0.00021223817998900448, + "loss": 3.4897, + "step": 44400 + }, + { + "epoch": 0.445, + "grad_norm": 0.08925904333591461, + "learning_rate": 0.000211999576001272, + "loss": 3.4917, + "step": 44500 + }, + { + "epoch": 0.446, + "grad_norm": 0.09652265906333923, + "learning_rate": 0.0002117617749438134, + "loss": 3.4887, + "step": 44600 + }, + { + "epoch": 0.447, + "grad_norm": 0.087586909532547, + "learning_rate": 0.0002115247723234508, + "loss": 3.4895, + "step": 44700 + }, + { + "epoch": 0.448, + "grad_norm": 0.11652135103940964, + "learning_rate": 0.00021128856368212914, + "loss": 3.4892, + "step": 44800 + }, + { + "epoch": 0.449, + "grad_norm": 0.10191888362169266, + "learning_rate": 0.00021105314459656365, + "loss": 3.4888, + "step": 44900 + }, + { + "epoch": 0.45, + "grad_norm": 0.09599770605564117, + "learning_rate": 0.00021081851067789197, + "loss": 3.4892, + "step": 45000 + }, + { + "epoch": 0.45, + "eval_loss": 3.91796875, + "eval_runtime": 33.1684, + "eval_samples_per_second": 18155.468, + "eval_steps_per_second": 17.758, + "step": 45000 + }, + { + "epoch": 0.451, + "grad_norm": 0.11355946213006973, + "learning_rate": 0.00021058465757133065, + "loss": 3.4885, + "step": 45100 + }, + { + "epoch": 0.452, + "grad_norm": 0.10500317811965942, + "learning_rate": 0.0002103515809558356, + "loss": 3.4864, + "step": 45200 + }, + { + "epoch": 0.453, + "grad_norm": 0.12674091756343842, + "learning_rate": 0.00021011927654376682, + "loss": 3.4877, + "step": 45300 + }, + { + "epoch": 0.454, + "grad_norm": 0.09490884095430374, + "learning_rate": 0.00020988774008055678, + "loss": 3.4868, + "step": 45400 + }, + { + "epoch": 0.455, + "grad_norm": 0.0876164361834526, + "learning_rate": 0.00020965696734438367, + "loss": 3.4869, + "step": 45500 + }, + { + "epoch": 0.456, + "grad_norm": 0.1012013703584671, + "learning_rate": 0.00020942695414584775, + "loss": 3.4887, + "step": 45600 + }, + { + "epoch": 0.457, + "grad_norm": 0.09138456732034683, + "learning_rate": 0.00020919769632765198, + "loss": 3.4879, + "step": 45700 + }, + { + "epoch": 0.458, + "grad_norm": 0.09227514266967773, + "learning_rate": 0.00020896918976428644, + "loss": 3.4866, + "step": 45800 + }, + { + "epoch": 0.459, + "grad_norm": 0.11846242845058441, + "learning_rate": 0.0002087414303617165, + "loss": 3.4861, + "step": 45900 + }, + { + "epoch": 0.46, + "grad_norm": 0.09923862665891647, + "learning_rate": 0.00020851441405707476, + "loss": 3.4869, + "step": 46000 + }, + { + "epoch": 0.46, + "eval_loss": 3.92578125, + "eval_runtime": 33.1689, + "eval_samples_per_second": 18155.218, + "eval_steps_per_second": 17.758, + "step": 46000 + }, + { + "epoch": 0.461, + "grad_norm": 0.11270363628864288, + "learning_rate": 0.00020828813681835672, + "loss": 3.4865, + "step": 46100 + }, + { + "epoch": 0.462, + "grad_norm": 0.09331326186656952, + "learning_rate": 0.00020806259464411974, + "loss": 3.4853, + "step": 46200 + }, + { + "epoch": 0.463, + "grad_norm": 0.09471999108791351, + "learning_rate": 0.00020783778356318634, + "loss": 3.4855, + "step": 46300 + }, + { + "epoch": 0.464, + "grad_norm": 0.10036460310220718, + "learning_rate": 0.00020761369963434993, + "loss": 3.486, + "step": 46400 + }, + { + "epoch": 0.465, + "grad_norm": 0.09839966893196106, + "learning_rate": 0.00020739033894608504, + "loss": 3.4836, + "step": 46500 + }, + { + "epoch": 0.466, + "grad_norm": 0.13888099789619446, + "learning_rate": 0.00020716769761626043, + "loss": 3.4845, + "step": 46600 + }, + { + "epoch": 0.467, + "grad_norm": 0.11780879646539688, + "learning_rate": 0.00020694577179185557, + "loss": 3.484, + "step": 46700 + }, + { + "epoch": 0.468, + "grad_norm": 0.10853921622037888, + "learning_rate": 0.0002067245576486808, + "loss": 3.4852, + "step": 46800 + }, + { + "epoch": 0.469, + "grad_norm": 0.10998079925775528, + "learning_rate": 0.00020650405139110026, + "loss": 3.483, + "step": 46900 + }, + { + "epoch": 0.47, + "grad_norm": 0.09292669594287872, + "learning_rate": 0.00020628424925175868, + "loss": 3.4847, + "step": 47000 + }, + { + "epoch": 0.47, + "eval_loss": 3.912109375, + "eval_runtime": 33.15, + "eval_samples_per_second": 18165.566, + "eval_steps_per_second": 17.768, + "step": 47000 + }, + { + "epoch": 0.471, + "grad_norm": 0.09031188488006592, + "learning_rate": 0.0002060651474913109, + "loss": 3.4842, + "step": 47100 + }, + { + "epoch": 0.472, + "grad_norm": 0.08594627678394318, + "learning_rate": 0.00020584674239815457, + "loss": 3.4835, + "step": 47200 + }, + { + "epoch": 0.473, + "grad_norm": 0.09004069864749908, + "learning_rate": 0.00020562903028816627, + "loss": 3.4846, + "step": 47300 + }, + { + "epoch": 0.474, + "grad_norm": 0.10319007933139801, + "learning_rate": 0.00020541200750444027, + "loss": 3.4859, + "step": 47400 + }, + { + "epoch": 0.475, + "grad_norm": 0.08985935151576996, + "learning_rate": 0.00020519567041703084, + "loss": 3.4808, + "step": 47500 + }, + { + "epoch": 0.476, + "grad_norm": 0.09091832488775253, + "learning_rate": 0.00020498001542269695, + "loss": 3.4846, + "step": 47600 + }, + { + "epoch": 0.477, + "grad_norm": 0.10014614462852478, + "learning_rate": 0.00020476503894465038, + "loss": 3.4818, + "step": 47700 + }, + { + "epoch": 0.478, + "grad_norm": 0.11294897645711899, + "learning_rate": 0.00020455073743230656, + "loss": 3.4842, + "step": 47800 + }, + { + "epoch": 0.479, + "grad_norm": 0.10061386972665787, + "learning_rate": 0.0002043371073610381, + "loss": 3.4825, + "step": 47900 + }, + { + "epoch": 0.48, + "grad_norm": 0.09785416722297668, + "learning_rate": 0.00020412414523193154, + "loss": 3.4838, + "step": 48000 + }, + { + "epoch": 0.48, + "eval_loss": 3.916015625, + "eval_runtime": 32.9883, + "eval_samples_per_second": 18254.585, + "eval_steps_per_second": 17.855, + "step": 48000 + }, + { + "epoch": 0.481, + "grad_norm": 0.09520223736763, + "learning_rate": 0.0002039118475715464, + "loss": 3.485, + "step": 48100 + }, + { + "epoch": 0.482, + "grad_norm": 0.0920218825340271, + "learning_rate": 0.00020370021093167763, + "loss": 3.4828, + "step": 48200 + }, + { + "epoch": 0.483, + "grad_norm": 0.12049808353185654, + "learning_rate": 0.00020348923188911992, + "loss": 3.4809, + "step": 48300 + }, + { + "epoch": 0.484, + "grad_norm": 0.11325296759605408, + "learning_rate": 0.00020327890704543544, + "loss": 3.4795, + "step": 48400 + }, + { + "epoch": 0.485, + "grad_norm": 0.0982460081577301, + "learning_rate": 0.00020306923302672384, + "loss": 3.4791, + "step": 48500 + }, + { + "epoch": 0.486, + "grad_norm": 0.10175520926713943, + "learning_rate": 0.00020286020648339485, + "loss": 3.4821, + "step": 48600 + }, + { + "epoch": 0.487, + "grad_norm": 0.10392586141824722, + "learning_rate": 0.00020265182408994378, + "loss": 3.4809, + "step": 48700 + }, + { + "epoch": 0.488, + "grad_norm": 0.08581174165010452, + "learning_rate": 0.000202444082544729, + "loss": 3.4805, + "step": 48800 + }, + { + "epoch": 0.489, + "grad_norm": 0.0906348004937172, + "learning_rate": 0.0002022369785697524, + "loss": 3.4827, + "step": 48900 + }, + { + "epoch": 0.49, + "grad_norm": 0.11659245938062668, + "learning_rate": 0.00020203050891044215, + "loss": 3.4832, + "step": 49000 + }, + { + "epoch": 0.49, + "eval_loss": 3.912109375, + "eval_runtime": 33.0, + "eval_samples_per_second": 18248.129, + "eval_steps_per_second": 17.848, + "step": 49000 + }, + { + "epoch": 0.491, + "grad_norm": 0.0962797999382019, + "learning_rate": 0.00020182467033543782, + "loss": 3.4812, + "step": 49100 + }, + { + "epoch": 0.492, + "grad_norm": 0.1375749707221985, + "learning_rate": 0.00020161945963637794, + "loss": 3.4795, + "step": 49200 + }, + { + "epoch": 0.493, + "grad_norm": 0.10032547265291214, + "learning_rate": 0.0002014148736276902, + "loss": 3.4767, + "step": 49300 + }, + { + "epoch": 0.494, + "grad_norm": 0.11045530438423157, + "learning_rate": 0.00020121090914638343, + "loss": 3.4803, + "step": 49400 + }, + { + "epoch": 0.495, + "grad_norm": 0.10294586420059204, + "learning_rate": 0.00020100756305184244, + "loss": 3.4789, + "step": 49500 + }, + { + "epoch": 0.496, + "grad_norm": 0.09476403146982193, + "learning_rate": 0.00020080483222562474, + "loss": 3.4796, + "step": 49600 + }, + { + "epoch": 0.497, + "grad_norm": 0.11336295306682587, + "learning_rate": 0.00020060271357125986, + "loss": 3.4812, + "step": 49700 + }, + { + "epoch": 0.498, + "grad_norm": 0.11719685047864914, + "learning_rate": 0.0002004012040140506, + "loss": 3.4772, + "step": 49800 + }, + { + "epoch": 0.499, + "grad_norm": 0.08824979513883591, + "learning_rate": 0.00020020030050087656, + "loss": 3.4779, + "step": 49900 + }, + { + "epoch": 0.5, + "grad_norm": 0.09348143637180328, + "learning_rate": 0.0002, + "loss": 3.4794, + "step": 50000 + }, + { + "epoch": 0.5, + "eval_loss": 3.90625, + "eval_runtime": 33.1492, + "eval_samples_per_second": 18165.994, + "eval_steps_per_second": 17.768, + "step": 50000 + }, + { + "epoch": 0.501, + "grad_norm": 0.10792515426874161, + "learning_rate": 0.00019980029950087343, + "loss": 3.4794, + "step": 50100 + }, + { + "epoch": 0.502, + "grad_norm": 0.09229454398155212, + "learning_rate": 0.0001996011960139498, + "loss": 3.4782, + "step": 50200 + }, + { + "epoch": 0.503, + "grad_norm": 0.09463869780302048, + "learning_rate": 0.00019940268657049439, + "loss": 3.4774, + "step": 50300 + }, + { + "epoch": 0.504, + "grad_norm": 0.09181378781795502, + "learning_rate": 0.00019920476822239894, + "loss": 3.4771, + "step": 50400 + }, + { + "epoch": 0.505, + "grad_norm": 0.09562738984823227, + "learning_rate": 0.00019900743804199784, + "loss": 3.4781, + "step": 50500 + }, + { + "epoch": 0.506, + "grad_norm": 0.08760121464729309, + "learning_rate": 0.000198810693121886, + "loss": 3.478, + "step": 50600 + }, + { + "epoch": 0.507, + "grad_norm": 0.09761682897806168, + "learning_rate": 0.00019861453057473934, + "loss": 3.4747, + "step": 50700 + }, + { + "epoch": 0.508, + "grad_norm": 0.09359412640333176, + "learning_rate": 0.00019841894753313627, + "loss": 3.4773, + "step": 50800 + }, + { + "epoch": 0.509, + "grad_norm": 0.11783072352409363, + "learning_rate": 0.00019822394114938217, + "loss": 3.4758, + "step": 50900 + }, + { + "epoch": 0.51, + "grad_norm": 0.09198185056447983, + "learning_rate": 0.00019802950859533488, + "loss": 3.4783, + "step": 51000 + }, + { + "epoch": 0.51, + "eval_loss": 3.912109375, + "eval_runtime": 32.9883, + "eval_samples_per_second": 18254.62, + "eval_steps_per_second": 17.855, + "step": 51000 + }, + { + "epoch": 0.511, + "grad_norm": 0.11417073011398315, + "learning_rate": 0.00019783564706223267, + "loss": 3.4765, + "step": 51100 + }, + { + "epoch": 0.512, + "grad_norm": 0.09531737864017487, + "learning_rate": 0.00019764235376052369, + "loss": 3.477, + "step": 51200 + }, + { + "epoch": 0.513, + "grad_norm": 0.11220179498195648, + "learning_rate": 0.00019744962591969746, + "loss": 3.4757, + "step": 51300 + }, + { + "epoch": 0.514, + "grad_norm": 0.09061618894338608, + "learning_rate": 0.0001972574607881179, + "loss": 3.475, + "step": 51400 + }, + { + "epoch": 0.515, + "grad_norm": 0.0985003113746643, + "learning_rate": 0.00019706585563285863, + "loss": 3.4732, + "step": 51500 + }, + { + "epoch": 0.516, + "grad_norm": 0.10033012181520462, + "learning_rate": 0.00019687480773953944, + "loss": 3.474, + "step": 51600 + }, + { + "epoch": 0.517, + "grad_norm": 0.09861289709806442, + "learning_rate": 0.00019668431441216495, + "loss": 3.473, + "step": 51700 + }, + { + "epoch": 0.518, + "grad_norm": 0.13850615918636322, + "learning_rate": 0.00019649437297296484, + "loss": 3.4756, + "step": 51800 + }, + { + "epoch": 0.519, + "grad_norm": 0.12541039288043976, + "learning_rate": 0.00019630498076223554, + "loss": 3.4747, + "step": 51900 + }, + { + "epoch": 0.52, + "grad_norm": 0.09361404925584793, + "learning_rate": 0.00019611613513818404, + "loss": 3.4794, + "step": 52000 + }, + { + "epoch": 0.52, + "eval_loss": 3.9140625, + "eval_runtime": 32.9814, + "eval_samples_per_second": 18258.432, + "eval_steps_per_second": 17.859, + "step": 52000 + }, + { + "epoch": 0.521, + "grad_norm": 0.09503428637981415, + "learning_rate": 0.00019592783347677307, + "loss": 3.4736, + "step": 52100 + }, + { + "epoch": 0.522, + "grad_norm": 0.10571688413619995, + "learning_rate": 0.00019574007317156784, + "loss": 3.4746, + "step": 52200 + }, + { + "epoch": 0.523, + "grad_norm": 0.12886875867843628, + "learning_rate": 0.00019555285163358466, + "loss": 3.4716, + "step": 52300 + }, + { + "epoch": 0.524, + "grad_norm": 0.11490379273891449, + "learning_rate": 0.00019536616629114086, + "loss": 3.4764, + "step": 52400 + }, + { + "epoch": 0.525, + "grad_norm": 0.09640925377607346, + "learning_rate": 0.00019518001458970665, + "loss": 3.4749, + "step": 52500 + }, + { + "epoch": 0.526, + "grad_norm": 0.11207705736160278, + "learning_rate": 0.00019499439399175798, + "loss": 3.4747, + "step": 52600 + }, + { + "epoch": 0.527, + "grad_norm": 0.10838747769594193, + "learning_rate": 0.00019480930197663147, + "loss": 3.4717, + "step": 52700 + }, + { + "epoch": 0.528, + "grad_norm": 0.11686636507511139, + "learning_rate": 0.00019462473604038076, + "loss": 3.4725, + "step": 52800 + }, + { + "epoch": 0.529, + "grad_norm": 0.0926726907491684, + "learning_rate": 0.00019444069369563388, + "loss": 3.4735, + "step": 52900 + }, + { + "epoch": 0.53, + "grad_norm": 0.08765498548746109, + "learning_rate": 0.00019425717247145284, + "loss": 3.4718, + "step": 53000 + }, + { + "epoch": 0.53, + "eval_loss": 3.912109375, + "eval_runtime": 32.9799, + "eval_samples_per_second": 18259.264, + "eval_steps_per_second": 17.859, + "step": 53000 + }, + { + "epoch": 0.531, + "grad_norm": 0.0892767682671547, + "learning_rate": 0.000194074169913194, + "loss": 3.4701, + "step": 53100 + }, + { + "epoch": 0.532, + "grad_norm": 0.10058142244815826, + "learning_rate": 0.0001938916835823703, + "loss": 3.4715, + "step": 53200 + }, + { + "epoch": 0.533, + "grad_norm": 0.09021036326885223, + "learning_rate": 0.0001937097110565149, + "loss": 3.4736, + "step": 53300 + }, + { + "epoch": 0.534, + "grad_norm": 0.10363597422838211, + "learning_rate": 0.00019352824992904588, + "loss": 3.4711, + "step": 53400 + }, + { + "epoch": 0.535, + "grad_norm": 0.08915389329195023, + "learning_rate": 0.0001933472978091327, + "loss": 3.4729, + "step": 53500 + }, + { + "epoch": 0.536, + "grad_norm": 0.1250082552433014, + "learning_rate": 0.00019316685232156395, + "loss": 3.4727, + "step": 53600 + }, + { + "epoch": 0.537, + "grad_norm": 0.10531096905469894, + "learning_rate": 0.00019298691110661623, + "loss": 3.4734, + "step": 53700 + }, + { + "epoch": 0.538, + "grad_norm": 0.10072151571512222, + "learning_rate": 0.00019280747181992476, + "loss": 3.472, + "step": 53800 + }, + { + "epoch": 0.539, + "grad_norm": 0.08304274082183838, + "learning_rate": 0.0001926285321323549, + "loss": 3.4698, + "step": 53900 + }, + { + "epoch": 0.54, + "grad_norm": 0.08655515313148499, + "learning_rate": 0.00019245008972987527, + "loss": 3.4707, + "step": 54000 + }, + { + "epoch": 0.54, + "eval_loss": 3.904296875, + "eval_runtime": 32.9698, + "eval_samples_per_second": 18264.853, + "eval_steps_per_second": 17.865, + "step": 54000 + }, + { + "epoch": 0.541, + "grad_norm": 0.09470842778682709, + "learning_rate": 0.00019227214231343207, + "loss": 3.4683, + "step": 54100 + }, + { + "epoch": 0.542, + "grad_norm": 0.09508795291185379, + "learning_rate": 0.00019209468759882463, + "loss": 3.4687, + "step": 54200 + }, + { + "epoch": 0.543, + "grad_norm": 0.09284261614084244, + "learning_rate": 0.00019191772331658236, + "loss": 3.4708, + "step": 54300 + }, + { + "epoch": 0.544, + "grad_norm": 0.09701448678970337, + "learning_rate": 0.0001917412472118426, + "loss": 3.4713, + "step": 54400 + }, + { + "epoch": 0.545, + "grad_norm": 0.11029274761676788, + "learning_rate": 0.0001915652570442303, + "loss": 3.4722, + "step": 54500 + }, + { + "epoch": 0.546, + "grad_norm": 0.08805020898580551, + "learning_rate": 0.00019138975058773818, + "loss": 3.4685, + "step": 54600 + }, + { + "epoch": 0.547, + "grad_norm": 0.09119655936956406, + "learning_rate": 0.00019121472563060887, + "loss": 3.4686, + "step": 54700 + }, + { + "epoch": 0.548, + "grad_norm": 0.09098175168037415, + "learning_rate": 0.00019104017997521752, + "loss": 3.4695, + "step": 54800 + }, + { + "epoch": 0.549, + "grad_norm": 0.10574721544981003, + "learning_rate": 0.00019086611143795607, + "loss": 3.4681, + "step": 54900 + }, + { + "epoch": 0.55, + "grad_norm": 0.09101998060941696, + "learning_rate": 0.00019069251784911847, + "loss": 3.4679, + "step": 55000 + }, + { + "epoch": 0.55, + "eval_loss": 3.90625, + "eval_runtime": 32.9737, + "eval_samples_per_second": 18262.678, + "eval_steps_per_second": 17.863, + "step": 55000 + }, + { + "epoch": 0.551, + "grad_norm": 0.09237556904554367, + "learning_rate": 0.00019051939705278708, + "loss": 3.4695, + "step": 55100 + }, + { + "epoch": 0.552, + "grad_norm": 0.0976092666387558, + "learning_rate": 0.00019034674690672024, + "loss": 3.4685, + "step": 55200 + }, + { + "epoch": 0.553, + "grad_norm": 0.09669974446296692, + "learning_rate": 0.00019017456528224098, + "loss": 3.4668, + "step": 55300 + }, + { + "epoch": 0.554, + "grad_norm": 0.08793047815561295, + "learning_rate": 0.0001900028500641266, + "loss": 3.4681, + "step": 55400 + }, + { + "epoch": 0.555, + "grad_norm": 0.12426000833511353, + "learning_rate": 0.0001898315991504998, + "loss": 3.4674, + "step": 55500 + }, + { + "epoch": 0.556, + "grad_norm": 0.10039014369249344, + "learning_rate": 0.0001896608104527204, + "loss": 3.4688, + "step": 55600 + }, + { + "epoch": 0.557, + "grad_norm": 0.10602252185344696, + "learning_rate": 0.00018949048189527846, + "loss": 3.4674, + "step": 55700 + }, + { + "epoch": 0.558, + "grad_norm": 0.13479745388031006, + "learning_rate": 0.00018932061141568827, + "loss": 3.4679, + "step": 55800 + }, + { + "epoch": 0.559, + "grad_norm": 0.08916186541318893, + "learning_rate": 0.0001891511969643836, + "loss": 3.467, + "step": 55900 + }, + { + "epoch": 0.56, + "grad_norm": 0.12169528752565384, + "learning_rate": 0.0001889822365046136, + "loss": 3.4648, + "step": 56000 + }, + { + "epoch": 0.56, + "eval_loss": 3.90234375, + "eval_runtime": 33.0103, + "eval_samples_per_second": 18242.4, + "eval_steps_per_second": 17.843, + "step": 56000 + }, + { + "epoch": 0.561, + "grad_norm": 0.0936431810259819, + "learning_rate": 0.00018881372801234024, + "loss": 3.4669, + "step": 56100 + }, + { + "epoch": 0.562, + "grad_norm": 0.100394107401371, + "learning_rate": 0.00018864566947613624, + "loss": 3.4661, + "step": 56200 + }, + { + "epoch": 0.563, + "grad_norm": 0.0859285295009613, + "learning_rate": 0.00018847805889708435, + "loss": 3.4661, + "step": 56300 + }, + { + "epoch": 0.564, + "grad_norm": 0.08721830695867538, + "learning_rate": 0.00018831089428867736, + "loss": 3.4658, + "step": 56400 + }, + { + "epoch": 0.565, + "grad_norm": 0.15624579787254333, + "learning_rate": 0.00018814417367671947, + "loss": 3.4653, + "step": 56500 + }, + { + "epoch": 0.566, + "grad_norm": 0.09214714914560318, + "learning_rate": 0.00018797789509922812, + "loss": 3.4642, + "step": 56600 + }, + { + "epoch": 0.567, + "grad_norm": 0.13581374287605286, + "learning_rate": 0.000187812056606337, + "loss": 3.4637, + "step": 56700 + }, + { + "epoch": 0.568, + "grad_norm": 0.09718205779790878, + "learning_rate": 0.0001876466562602004, + "loss": 3.4657, + "step": 56800 + }, + { + "epoch": 0.569, + "grad_norm": 0.09438052028417587, + "learning_rate": 0.00018748169213489755, + "loss": 3.4671, + "step": 56900 + }, + { + "epoch": 0.57, + "grad_norm": 0.090947225689888, + "learning_rate": 0.0001873171623163388, + "loss": 3.4655, + "step": 57000 + }, + { + "epoch": 0.57, + "eval_loss": 3.900390625, + "eval_runtime": 32.9917, + "eval_samples_per_second": 18252.689, + "eval_steps_per_second": 17.853, + "step": 57000 + }, + { + "epoch": 0.571, + "grad_norm": 0.09600440412759781, + "learning_rate": 0.0001871530649021723, + "loss": 3.4676, + "step": 57100 + }, + { + "epoch": 0.572, + "grad_norm": 0.09546720236539841, + "learning_rate": 0.00018698939800169143, + "loss": 3.4651, + "step": 57200 + }, + { + "epoch": 0.573, + "grad_norm": 0.11584016680717468, + "learning_rate": 0.0001868261597357436, + "loss": 3.4654, + "step": 57300 + }, + { + "epoch": 0.574, + "grad_norm": 0.09096308052539825, + "learning_rate": 0.00018666334823663937, + "loss": 3.4642, + "step": 57400 + }, + { + "epoch": 0.575, + "grad_norm": 0.09048023074865341, + "learning_rate": 0.00018650096164806276, + "loss": 3.4632, + "step": 57500 + }, + { + "epoch": 0.576, + "grad_norm": 0.08897216618061066, + "learning_rate": 0.0001863389981249825, + "loss": 3.4642, + "step": 57600 + }, + { + "epoch": 0.577, + "grad_norm": 0.08574885874986649, + "learning_rate": 0.00018617745583356373, + "loss": 3.4645, + "step": 57700 + }, + { + "epoch": 0.578, + "grad_norm": 0.09016621857881546, + "learning_rate": 0.00018601633295108115, + "loss": 3.4651, + "step": 57800 + }, + { + "epoch": 0.579, + "grad_norm": 0.08941857516765594, + "learning_rate": 0.0001858556276658322, + "loss": 3.4656, + "step": 57900 + }, + { + "epoch": 0.58, + "grad_norm": 0.08731842786073685, + "learning_rate": 0.00018569533817705186, + "loss": 3.4664, + "step": 58000 + }, + { + "epoch": 0.58, + "eval_loss": 3.896484375, + "eval_runtime": 32.9962, + "eval_samples_per_second": 18250.233, + "eval_steps_per_second": 17.851, + "step": 58000 + }, + { + "epoch": 0.581, + "grad_norm": 0.09401644766330719, + "learning_rate": 0.00018553546269482776, + "loss": 3.462, + "step": 58100 + }, + { + "epoch": 0.582, + "grad_norm": 0.1261298507452011, + "learning_rate": 0.00018537599944001617, + "loss": 3.4631, + "step": 58200 + }, + { + "epoch": 0.583, + "grad_norm": 0.08777070790529251, + "learning_rate": 0.00018521694664415904, + "loss": 3.4647, + "step": 58300 + }, + { + "epoch": 0.584, + "grad_norm": 0.10590939223766327, + "learning_rate": 0.00018505830254940132, + "loss": 3.4645, + "step": 58400 + }, + { + "epoch": 0.585, + "grad_norm": 0.08615025877952576, + "learning_rate": 0.0001849000654084097, + "loss": 3.463, + "step": 58500 + }, + { + "epoch": 0.586, + "grad_norm": 0.09571292251348495, + "learning_rate": 0.00018474223348429158, + "loss": 3.4624, + "step": 58600 + }, + { + "epoch": 0.587, + "grad_norm": 0.08672121912240982, + "learning_rate": 0.000184584805050515, + "loss": 3.4636, + "step": 58700 + }, + { + "epoch": 0.588, + "grad_norm": 0.09722130000591278, + "learning_rate": 0.00018442777839082936, + "loss": 3.4615, + "step": 58800 + }, + { + "epoch": 0.589, + "grad_norm": 0.08235087245702744, + "learning_rate": 0.00018427115179918692, + "loss": 3.4631, + "step": 58900 + }, + { + "epoch": 0.59, + "grad_norm": 0.11260483413934708, + "learning_rate": 0.00018411492357966468, + "loss": 3.46, + "step": 59000 + }, + { + "epoch": 0.59, + "eval_loss": 3.8984375, + "eval_runtime": 32.986, + "eval_samples_per_second": 18255.851, + "eval_steps_per_second": 17.856, + "step": 59000 + }, + { + "epoch": 0.591, + "grad_norm": 0.09491027891635895, + "learning_rate": 0.00018395909204638758, + "loss": 3.4612, + "step": 59100 + }, + { + "epoch": 0.592, + "grad_norm": 0.09216572344303131, + "learning_rate": 0.00018380365552345197, + "loss": 3.4632, + "step": 59200 + }, + { + "epoch": 0.593, + "grad_norm": 0.09314817935228348, + "learning_rate": 0.00018364861234484967, + "loss": 3.4616, + "step": 59300 + }, + { + "epoch": 0.594, + "grad_norm": 0.10169938951730728, + "learning_rate": 0.00018349396085439343, + "loss": 3.4638, + "step": 59400 + }, + { + "epoch": 0.595, + "grad_norm": 0.10830460488796234, + "learning_rate": 0.00018333969940564226, + "loss": 3.4622, + "step": 59500 + }, + { + "epoch": 0.596, + "grad_norm": 0.09347419440746307, + "learning_rate": 0.00018318582636182794, + "loss": 3.461, + "step": 59600 + }, + { + "epoch": 0.597, + "grad_norm": 0.09264504164457321, + "learning_rate": 0.00018303234009578204, + "loss": 3.4604, + "step": 59700 + }, + { + "epoch": 0.598, + "grad_norm": 0.08877206593751907, + "learning_rate": 0.00018287923898986378, + "loss": 3.4622, + "step": 59800 + }, + { + "epoch": 0.599, + "grad_norm": 0.09566622227430344, + "learning_rate": 0.00018272652143588817, + "loss": 3.4635, + "step": 59900 + }, + { + "epoch": 0.6, + "grad_norm": 0.10193250328302383, + "learning_rate": 0.00018257418583505537, + "loss": 3.4608, + "step": 60000 + }, + { + "epoch": 0.6, + "eval_loss": 3.8984375, + "eval_runtime": 32.9947, + "eval_samples_per_second": 18251.046, + "eval_steps_per_second": 17.851, + "step": 60000 + }, + { + "epoch": 0.601, + "grad_norm": 0.09149635583162308, + "learning_rate": 0.00018242223059788013, + "loss": 3.4603, + "step": 60100 + }, + { + "epoch": 0.602, + "grad_norm": 0.09358430653810501, + "learning_rate": 0.0001822706541441223, + "loss": 3.4595, + "step": 60200 + }, + { + "epoch": 0.603, + "grad_norm": 0.10326792299747467, + "learning_rate": 0.00018211945490271766, + "loss": 3.4619, + "step": 60300 + }, + { + "epoch": 0.604, + "grad_norm": 0.10516923666000366, + "learning_rate": 0.00018196863131170974, + "loss": 3.4592, + "step": 60400 + }, + { + "epoch": 0.605, + "grad_norm": 0.09201698005199432, + "learning_rate": 0.00018181818181818183, + "loss": 3.4594, + "step": 60500 + }, + { + "epoch": 0.606, + "grad_norm": 0.12762495875358582, + "learning_rate": 0.00018166810487818988, + "loss": 3.4608, + "step": 60600 + }, + { + "epoch": 0.607, + "grad_norm": 0.09665093570947647, + "learning_rate": 0.00018151839895669614, + "loss": 3.4579, + "step": 60700 + }, + { + "epoch": 0.608, + "grad_norm": 0.0870705246925354, + "learning_rate": 0.00018136906252750294, + "loss": 3.4588, + "step": 60800 + }, + { + "epoch": 0.609, + "grad_norm": 0.0998244658112526, + "learning_rate": 0.00018122009407318745, + "loss": 3.4589, + "step": 60900 + }, + { + "epoch": 0.61, + "grad_norm": 0.08676362037658691, + "learning_rate": 0.00018107149208503707, + "loss": 3.4604, + "step": 61000 + }, + { + "epoch": 0.61, + "eval_loss": 3.892578125, + "eval_runtime": 33.0945, + "eval_samples_per_second": 18195.999, + "eval_steps_per_second": 17.798, + "step": 61000 + }, + { + "epoch": 0.611, + "grad_norm": 0.09438201040029526, + "learning_rate": 0.000180923255062985, + "loss": 3.4588, + "step": 61100 + }, + { + "epoch": 0.612, + "grad_norm": 0.09467477351427078, + "learning_rate": 0.0001807753815155468, + "loss": 3.4595, + "step": 61200 + }, + { + "epoch": 0.613, + "grad_norm": 0.09366316348314285, + "learning_rate": 0.00018062786995975738, + "loss": 3.4596, + "step": 61300 + }, + { + "epoch": 0.614, + "grad_norm": 0.09797662496566772, + "learning_rate": 0.0001804807189211084, + "loss": 3.46, + "step": 61400 + }, + { + "epoch": 0.615, + "grad_norm": 0.09239408373832703, + "learning_rate": 0.00018033392693348645, + "loss": 3.4592, + "step": 61500 + }, + { + "epoch": 0.616, + "grad_norm": 0.1170559823513031, + "learning_rate": 0.0001801874925391118, + "loss": 3.4594, + "step": 61600 + }, + { + "epoch": 0.617, + "grad_norm": 0.11294198781251907, + "learning_rate": 0.00018004141428847736, + "loss": 3.4566, + "step": 61700 + }, + { + "epoch": 0.618, + "grad_norm": 0.08425901085138321, + "learning_rate": 0.00017989569074028863, + "loss": 3.4582, + "step": 61800 + }, + { + "epoch": 0.619, + "grad_norm": 0.08901657909154892, + "learning_rate": 0.0001797503204614039, + "loss": 3.4575, + "step": 61900 + }, + { + "epoch": 0.62, + "grad_norm": 0.09452513605356216, + "learning_rate": 0.00017960530202677493, + "loss": 3.4584, + "step": 62000 + }, + { + "epoch": 0.62, + "eval_loss": 3.888671875, + "eval_runtime": 33.0119, + "eval_samples_per_second": 18241.555, + "eval_steps_per_second": 17.842, + "step": 62000 + }, + { + "epoch": 0.621, + "grad_norm": 0.10306137800216675, + "learning_rate": 0.0001794606340193885, + "loss": 3.4585, + "step": 62100 + }, + { + "epoch": 0.622, + "grad_norm": 0.09906180948019028, + "learning_rate": 0.00017931631503020814, + "loss": 3.4559, + "step": 62200 + }, + { + "epoch": 0.623, + "grad_norm": 0.09659173339605331, + "learning_rate": 0.00017917234365811654, + "loss": 3.4568, + "step": 62300 + }, + { + "epoch": 0.624, + "grad_norm": 0.08893030881881714, + "learning_rate": 0.0001790287185098582, + "loss": 3.4596, + "step": 62400 + }, + { + "epoch": 0.625, + "grad_norm": 0.09648360311985016, + "learning_rate": 0.00017888543819998318, + "loss": 3.4573, + "step": 62500 + }, + { + "epoch": 0.626, + "grad_norm": 0.08842136710882187, + "learning_rate": 0.0001787425013507906, + "loss": 3.4575, + "step": 62600 + }, + { + "epoch": 0.627, + "grad_norm": 0.097404845058918, + "learning_rate": 0.00017859990659227327, + "loss": 3.4553, + "step": 62700 + }, + { + "epoch": 0.628, + "grad_norm": 0.10879901051521301, + "learning_rate": 0.0001784576525620624, + "loss": 3.4557, + "step": 62800 + }, + { + "epoch": 0.629, + "grad_norm": 0.08811251074075699, + "learning_rate": 0.000178315737905373, + "loss": 3.4547, + "step": 62900 + }, + { + "epoch": 0.63, + "grad_norm": 0.09087971597909927, + "learning_rate": 0.0001781741612749496, + "loss": 3.4556, + "step": 63000 + }, + { + "epoch": 0.63, + "eval_loss": 3.89453125, + "eval_runtime": 33.0103, + "eval_samples_per_second": 18242.415, + "eval_steps_per_second": 17.843, + "step": 63000 + }, + { + "epoch": 0.631, + "grad_norm": 0.08818445354700089, + "learning_rate": 0.0001780329213310126, + "loss": 3.4553, + "step": 63100 + }, + { + "epoch": 0.632, + "grad_norm": 0.09016218781471252, + "learning_rate": 0.000177892016741205, + "loss": 3.4546, + "step": 63200 + }, + { + "epoch": 0.633, + "grad_norm": 0.08760054409503937, + "learning_rate": 0.0001777514461805397, + "loss": 3.4548, + "step": 63300 + }, + { + "epoch": 0.634, + "grad_norm": 0.08488897979259491, + "learning_rate": 0.00017761120833134698, + "loss": 3.4575, + "step": 63400 + }, + { + "epoch": 0.635, + "grad_norm": 0.09008808434009552, + "learning_rate": 0.00017747130188322277, + "loss": 3.4554, + "step": 63500 + }, + { + "epoch": 0.636, + "grad_norm": 0.09507942944765091, + "learning_rate": 0.00017733172553297715, + "loss": 3.4548, + "step": 63600 + }, + { + "epoch": 0.637, + "grad_norm": 0.08736253529787064, + "learning_rate": 0.00017719247798458348, + "loss": 3.4547, + "step": 63700 + }, + { + "epoch": 0.638, + "grad_norm": 0.1137012243270874, + "learning_rate": 0.00017705355794912778, + "loss": 3.4586, + "step": 63800 + }, + { + "epoch": 0.639, + "grad_norm": 0.100893035531044, + "learning_rate": 0.00017691496414475844, + "loss": 3.4563, + "step": 63900 + }, + { + "epoch": 0.64, + "grad_norm": 0.0883665531873703, + "learning_rate": 0.00017677669529663688, + "loss": 3.4545, + "step": 64000 + }, + { + "epoch": 0.64, + "eval_loss": 3.90234375, + "eval_runtime": 32.9836, + "eval_samples_per_second": 18257.192, + "eval_steps_per_second": 17.857, + "step": 64000 + }, + { + "epoch": 0.641, + "grad_norm": 0.09608853608369827, + "learning_rate": 0.00017663875013688816, + "loss": 3.4548, + "step": 64100 + }, + { + "epoch": 0.642, + "grad_norm": 0.08302801102399826, + "learning_rate": 0.00017650112740455194, + "loss": 3.454, + "step": 64200 + }, + { + "epoch": 0.643, + "grad_norm": 0.08717923611402512, + "learning_rate": 0.0001763638258455345, + "loss": 3.4555, + "step": 64300 + }, + { + "epoch": 0.644, + "grad_norm": 0.11070983111858368, + "learning_rate": 0.00017622684421256035, + "loss": 3.4551, + "step": 64400 + }, + { + "epoch": 0.645, + "grad_norm": 0.10088195651769638, + "learning_rate": 0.00017609018126512477, + "loss": 3.4545, + "step": 64500 + }, + { + "epoch": 0.646, + "grad_norm": 0.08371099084615707, + "learning_rate": 0.00017595383576944672, + "loss": 3.4552, + "step": 64600 + }, + { + "epoch": 0.647, + "grad_norm": 0.08482113480567932, + "learning_rate": 0.00017581780649842193, + "loss": 3.4547, + "step": 64700 + }, + { + "epoch": 0.648, + "grad_norm": 0.10641901195049286, + "learning_rate": 0.00017568209223157663, + "loss": 3.4539, + "step": 64800 + }, + { + "epoch": 0.649, + "grad_norm": 0.12532317638397217, + "learning_rate": 0.00017554669175502143, + "loss": 3.4542, + "step": 64900 + }, + { + "epoch": 0.65, + "grad_norm": 0.10789231210947037, + "learning_rate": 0.00017541160386140586, + "loss": 3.4526, + "step": 65000 + }, + { + "epoch": 0.65, + "eval_loss": 3.884765625, + "eval_runtime": 32.991, + "eval_samples_per_second": 18253.083, + "eval_steps_per_second": 17.853, + "step": 65000 + }, + { + "epoch": 0.651, + "grad_norm": 0.10503752529621124, + "learning_rate": 0.00017527682734987297, + "loss": 3.4551, + "step": 65100 + }, + { + "epoch": 0.652, + "grad_norm": 0.08596997708082199, + "learning_rate": 0.00017514236102601468, + "loss": 3.4544, + "step": 65200 + }, + { + "epoch": 0.653, + "grad_norm": 0.10656549036502838, + "learning_rate": 0.00017500820370182729, + "loss": 3.452, + "step": 65300 + }, + { + "epoch": 0.654, + "grad_norm": 0.10545896738767624, + "learning_rate": 0.00017487435419566726, + "loss": 3.4525, + "step": 65400 + }, + { + "epoch": 0.655, + "grad_norm": 0.10238152742385864, + "learning_rate": 0.0001747408113322076, + "loss": 3.453, + "step": 65500 + }, + { + "epoch": 0.656, + "grad_norm": 0.1159452497959137, + "learning_rate": 0.00017460757394239458, + "loss": 3.4533, + "step": 65600 + }, + { + "epoch": 0.657, + "grad_norm": 0.08971402049064636, + "learning_rate": 0.00017447464086340456, + "loss": 3.4525, + "step": 65700 + }, + { + "epoch": 0.658, + "grad_norm": 0.09728628396987915, + "learning_rate": 0.00017434201093860166, + "loss": 3.4533, + "step": 65800 + }, + { + "epoch": 0.659, + "grad_norm": 0.10055437684059143, + "learning_rate": 0.00017420968301749517, + "loss": 3.4515, + "step": 65900 + }, + { + "epoch": 0.66, + "grad_norm": 0.10009339451789856, + "learning_rate": 0.00017407765595569785, + "loss": 3.4532, + "step": 66000 + }, + { + "epoch": 0.66, + "eval_loss": 3.888671875, + "eval_runtime": 33.007, + "eval_samples_per_second": 18244.268, + "eval_steps_per_second": 17.845, + "step": 66000 + }, + { + "epoch": 0.661, + "grad_norm": 0.08580023795366287, + "learning_rate": 0.00017394592861488424, + "loss": 3.4505, + "step": 66100 + }, + { + "epoch": 0.662, + "grad_norm": 0.08647996932268143, + "learning_rate": 0.00017381449986274955, + "loss": 3.4532, + "step": 66200 + }, + { + "epoch": 0.663, + "grad_norm": 0.08780326694250107, + "learning_rate": 0.00017368336857296875, + "loss": 3.4515, + "step": 66300 + }, + { + "epoch": 0.664, + "grad_norm": 0.09744898974895477, + "learning_rate": 0.00017355253362515583, + "loss": 3.4513, + "step": 66400 + }, + { + "epoch": 0.665, + "grad_norm": 0.10528256744146347, + "learning_rate": 0.000173421993904824, + "loss": 3.4534, + "step": 66500 + }, + { + "epoch": 0.666, + "grad_norm": 0.0887388065457344, + "learning_rate": 0.00017329174830334545, + "loss": 3.4532, + "step": 66600 + }, + { + "epoch": 0.667, + "grad_norm": 0.11578360199928284, + "learning_rate": 0.00017316179571791196, + "loss": 3.4502, + "step": 66700 + }, + { + "epoch": 0.668, + "grad_norm": 0.09660936146974564, + "learning_rate": 0.0001730321350514957, + "loss": 3.4531, + "step": 66800 + }, + { + "epoch": 0.669, + "grad_norm": 0.09446755796670914, + "learning_rate": 0.0001729027652128102, + "loss": 3.4543, + "step": 66900 + }, + { + "epoch": 0.67, + "grad_norm": 0.10190019011497498, + "learning_rate": 0.00017277368511627204, + "loss": 3.4509, + "step": 67000 + }, + { + "epoch": 0.67, + "eval_loss": 3.8828125, + "eval_runtime": 32.992, + "eval_samples_per_second": 18252.544, + "eval_steps_per_second": 17.853, + "step": 67000 + }, + { + "epoch": 0.671, + "grad_norm": 0.10701554268598557, + "learning_rate": 0.0001726448936819622, + "loss": 3.4513, + "step": 67100 + }, + { + "epoch": 0.672, + "grad_norm": 0.11148486286401749, + "learning_rate": 0.00017251638983558853, + "loss": 3.4505, + "step": 67200 + }, + { + "epoch": 0.673, + "grad_norm": 0.10375931113958359, + "learning_rate": 0.00017238817250844786, + "loss": 3.4514, + "step": 67300 + }, + { + "epoch": 0.674, + "grad_norm": 0.09482084214687347, + "learning_rate": 0.00017226024063738863, + "loss": 3.4498, + "step": 67400 + }, + { + "epoch": 0.675, + "grad_norm": 0.11015119403600693, + "learning_rate": 0.00017213259316477408, + "loss": 3.4503, + "step": 67500 + }, + { + "epoch": 0.676, + "grad_norm": 0.09139025211334229, + "learning_rate": 0.00017200522903844536, + "loss": 3.4503, + "step": 67600 + }, + { + "epoch": 0.677, + "grad_norm": 0.08138120919466019, + "learning_rate": 0.00017187814721168515, + "loss": 3.4496, + "step": 67700 + }, + { + "epoch": 0.678, + "grad_norm": 0.11369217187166214, + "learning_rate": 0.00017175134664318157, + "loss": 3.4493, + "step": 67800 + }, + { + "epoch": 0.679, + "grad_norm": 0.09752605855464935, + "learning_rate": 0.00017162482629699222, + "loss": 3.4484, + "step": 67900 + }, + { + "epoch": 0.68, + "grad_norm": 0.0917077511548996, + "learning_rate": 0.00017149858514250882, + "loss": 3.4493, + "step": 68000 + }, + { + "epoch": 0.68, + "eval_loss": 3.880859375, + "eval_runtime": 32.9961, + "eval_samples_per_second": 18250.289, + "eval_steps_per_second": 17.851, + "step": 68000 + }, + { + "epoch": 0.681, + "grad_norm": 0.09174978733062744, + "learning_rate": 0.00017137262215442185, + "loss": 3.45, + "step": 68100 + }, + { + "epoch": 0.682, + "grad_norm": 0.11849396675825119, + "learning_rate": 0.00017124693631268543, + "loss": 3.4511, + "step": 68200 + }, + { + "epoch": 0.683, + "grad_norm": 0.0938100814819336, + "learning_rate": 0.00017112152660248296, + "loss": 3.4496, + "step": 68300 + }, + { + "epoch": 0.684, + "grad_norm": 0.11769715696573257, + "learning_rate": 0.00017099639201419238, + "loss": 3.4495, + "step": 68400 + }, + { + "epoch": 0.685, + "grad_norm": 0.11145388334989548, + "learning_rate": 0.0001708715315433522, + "loss": 3.4493, + "step": 68500 + }, + { + "epoch": 0.686, + "grad_norm": 0.12704512476921082, + "learning_rate": 0.00017074694419062767, + "loss": 3.4503, + "step": 68600 + }, + { + "epoch": 0.687, + "grad_norm": 0.10009263455867767, + "learning_rate": 0.00017062262896177705, + "loss": 3.4527, + "step": 68700 + }, + { + "epoch": 0.688, + "grad_norm": 0.0949656069278717, + "learning_rate": 0.00017049858486761839, + "loss": 3.4497, + "step": 68800 + }, + { + "epoch": 0.689, + "grad_norm": 0.08776091784238815, + "learning_rate": 0.0001703748109239964, + "loss": 3.4493, + "step": 68900 + }, + { + "epoch": 0.69, + "grad_norm": 0.12844863533973694, + "learning_rate": 0.0001702513061517497, + "loss": 3.449, + "step": 69000 + }, + { + "epoch": 0.69, + "eval_loss": 3.8828125, + "eval_runtime": 33.0055, + "eval_samples_per_second": 18245.099, + "eval_steps_per_second": 17.846, + "step": 69000 + }, + { + "epoch": 0.691, + "grad_norm": 0.086809441447258, + "learning_rate": 0.0001701280695766784, + "loss": 3.4488, + "step": 69100 + }, + { + "epoch": 0.692, + "grad_norm": 0.09244389086961746, + "learning_rate": 0.0001700051002295115, + "loss": 3.4473, + "step": 69200 + }, + { + "epoch": 0.693, + "grad_norm": 0.08996045589447021, + "learning_rate": 0.00016988239714587518, + "loss": 3.4499, + "step": 69300 + }, + { + "epoch": 0.694, + "grad_norm": 0.10364508628845215, + "learning_rate": 0.00016975995936626098, + "loss": 3.4495, + "step": 69400 + }, + { + "epoch": 0.695, + "grad_norm": 0.09321583062410355, + "learning_rate": 0.0001696377859359942, + "loss": 3.4468, + "step": 69500 + }, + { + "epoch": 0.696, + "grad_norm": 0.09319949150085449, + "learning_rate": 0.0001695158759052026, + "loss": 3.447, + "step": 69600 + }, + { + "epoch": 0.697, + "grad_norm": 0.0915616974234581, + "learning_rate": 0.00016939422832878555, + "loss": 3.4461, + "step": 69700 + }, + { + "epoch": 0.698, + "grad_norm": 0.08925987035036087, + "learning_rate": 0.00016927284226638315, + "loss": 3.4479, + "step": 69800 + }, + { + "epoch": 0.699, + "grad_norm": 0.10480326414108276, + "learning_rate": 0.00016915171678234564, + "loss": 3.4477, + "step": 69900 + }, + { + "epoch": 0.7, + "grad_norm": 0.09662549942731857, + "learning_rate": 0.0001690308509457033, + "loss": 3.4464, + "step": 70000 + }, + { + "epoch": 0.7, + "eval_loss": 3.87890625, + "eval_runtime": 33.1572, + "eval_samples_per_second": 18161.615, + "eval_steps_per_second": 17.764, + "step": 70000 + }, + { + "epoch": 0.701, + "grad_norm": 0.08479616045951843, + "learning_rate": 0.0001689102438301362, + "loss": 3.4458, + "step": 70100 + }, + { + "epoch": 0.702, + "grad_norm": 0.09455426782369614, + "learning_rate": 0.00016878989451394444, + "loss": 3.447, + "step": 70200 + }, + { + "epoch": 0.703, + "grad_norm": 0.08542288839817047, + "learning_rate": 0.00016866980208001865, + "loss": 3.4462, + "step": 70300 + }, + { + "epoch": 0.704, + "grad_norm": 0.10905344784259796, + "learning_rate": 0.00016854996561581053, + "loss": 3.4454, + "step": 70400 + }, + { + "epoch": 0.705, + "grad_norm": 0.0927351713180542, + "learning_rate": 0.00016843038421330382, + "loss": 3.4444, + "step": 70500 + }, + { + "epoch": 0.706, + "grad_norm": 0.0991898775100708, + "learning_rate": 0.00016831105696898528, + "loss": 3.4476, + "step": 70600 + }, + { + "epoch": 0.707, + "grad_norm": 0.07934626936912537, + "learning_rate": 0.0001681919829838161, + "loss": 3.4465, + "step": 70700 + }, + { + "epoch": 0.708, + "grad_norm": 0.0833369791507721, + "learning_rate": 0.0001680731613632036, + "loss": 3.4482, + "step": 70800 + }, + { + "epoch": 0.709, + "grad_norm": 0.09017114341259003, + "learning_rate": 0.00016795459121697257, + "loss": 3.4453, + "step": 70900 + }, + { + "epoch": 0.71, + "grad_norm": 0.09019383788108826, + "learning_rate": 0.00016783627165933783, + "loss": 3.446, + "step": 71000 + }, + { + "epoch": 0.71, + "eval_loss": 3.890625, + "eval_runtime": 33.0104, + "eval_samples_per_second": 18242.397, + "eval_steps_per_second": 17.843, + "step": 71000 + }, + { + "epoch": 0.711, + "grad_norm": 0.09486021101474762, + "learning_rate": 0.0001677182018088759, + "loss": 3.4462, + "step": 71100 + }, + { + "epoch": 0.712, + "grad_norm": 0.09535174071788788, + "learning_rate": 0.00016760038078849775, + "loss": 3.4483, + "step": 71200 + }, + { + "epoch": 0.713, + "grad_norm": 0.08906789124011993, + "learning_rate": 0.00016748280772542137, + "loss": 3.445, + "step": 71300 + }, + { + "epoch": 0.714, + "grad_norm": 0.09585762023925781, + "learning_rate": 0.0001673654817511446, + "loss": 3.4469, + "step": 71400 + }, + { + "epoch": 0.715, + "grad_norm": 0.10558847337961197, + "learning_rate": 0.00016724840200141815, + "loss": 3.4462, + "step": 71500 + }, + { + "epoch": 0.716, + "grad_norm": 0.09401770681142807, + "learning_rate": 0.0001671315676162189, + "loss": 3.4436, + "step": 71600 + }, + { + "epoch": 0.717, + "grad_norm": 0.09375085681676865, + "learning_rate": 0.00016701497773972333, + "loss": 3.445, + "step": 71700 + }, + { + "epoch": 0.718, + "grad_norm": 0.09521766006946564, + "learning_rate": 0.00016689863152028126, + "loss": 3.443, + "step": 71800 + }, + { + "epoch": 0.719, + "grad_norm": 0.09246081113815308, + "learning_rate": 0.00016678252811038962, + "loss": 3.4458, + "step": 71900 + }, + { + "epoch": 0.72, + "grad_norm": 0.08197703957557678, + "learning_rate": 0.00016666666666666666, + "loss": 3.4456, + "step": 72000 + }, + { + "epoch": 0.72, + "eval_loss": 3.880859375, + "eval_runtime": 32.9885, + "eval_samples_per_second": 18254.48, + "eval_steps_per_second": 17.855, + "step": 72000 + }, + { + "epoch": 0.721, + "grad_norm": 0.08918161690235138, + "learning_rate": 0.00016655104634982608, + "loss": 3.4453, + "step": 72100 + }, + { + "epoch": 0.722, + "grad_norm": 0.08643663674592972, + "learning_rate": 0.00016643566632465153, + "loss": 3.4471, + "step": 72200 + }, + { + "epoch": 0.723, + "grad_norm": 0.08612331748008728, + "learning_rate": 0.0001663205257599714, + "loss": 3.4442, + "step": 72300 + }, + { + "epoch": 0.724, + "grad_norm": 0.11025555431842804, + "learning_rate": 0.00016620562382863342, + "loss": 3.4443, + "step": 72400 + }, + { + "epoch": 0.725, + "grad_norm": 0.09381308406591415, + "learning_rate": 0.00016609095970747994, + "loss": 3.4414, + "step": 72500 + }, + { + "epoch": 0.726, + "grad_norm": 0.094405896961689, + "learning_rate": 0.00016597653257732306, + "loss": 3.4438, + "step": 72600 + }, + { + "epoch": 0.727, + "grad_norm": 0.09081803262233734, + "learning_rate": 0.00016586234162292005, + "loss": 3.444, + "step": 72700 + }, + { + "epoch": 0.728, + "grad_norm": 0.11879288405179977, + "learning_rate": 0.00016574838603294897, + "loss": 3.4418, + "step": 72800 + }, + { + "epoch": 0.729, + "grad_norm": 0.09423980116844177, + "learning_rate": 0.00016563466499998442, + "loss": 3.4443, + "step": 72900 + }, + { + "epoch": 0.73, + "grad_norm": 0.0863361731171608, + "learning_rate": 0.00016552117772047358, + "loss": 3.4436, + "step": 73000 + }, + { + "epoch": 0.73, + "eval_loss": 3.884765625, + "eval_runtime": 33.0049, + "eval_samples_per_second": 18245.429, + "eval_steps_per_second": 17.846, + "step": 73000 + }, + { + "epoch": 0.731, + "grad_norm": 0.13333062827587128, + "learning_rate": 0.00016540792339471237, + "loss": 3.4447, + "step": 73100 + }, + { + "epoch": 0.732, + "grad_norm": 0.0957045704126358, + "learning_rate": 0.00016529490122682157, + "loss": 3.4425, + "step": 73200 + }, + { + "epoch": 0.733, + "grad_norm": 0.12515659630298615, + "learning_rate": 0.00016518211042472372, + "loss": 3.4423, + "step": 73300 + }, + { + "epoch": 0.734, + "grad_norm": 0.09493029862642288, + "learning_rate": 0.00016506955020011946, + "loss": 3.4401, + "step": 73400 + }, + { + "epoch": 0.735, + "grad_norm": 0.08881188184022903, + "learning_rate": 0.0001649572197684645, + "loss": 3.4419, + "step": 73500 + }, + { + "epoch": 0.736, + "grad_norm": 0.10456575453281403, + "learning_rate": 0.00016484511834894677, + "loss": 3.4428, + "step": 73600 + }, + { + "epoch": 0.737, + "grad_norm": 0.08246786147356033, + "learning_rate": 0.00016473324516446338, + "loss": 3.4445, + "step": 73700 + }, + { + "epoch": 0.738, + "grad_norm": 0.11760751903057098, + "learning_rate": 0.00016462159944159827, + "loss": 3.4431, + "step": 73800 + }, + { + "epoch": 0.739, + "grad_norm": 0.09637313336133957, + "learning_rate": 0.00016451018041059955, + "loss": 3.4426, + "step": 73900 + }, + { + "epoch": 0.74, + "grad_norm": 0.13058100640773773, + "learning_rate": 0.0001643989873053573, + "loss": 3.4425, + "step": 74000 + }, + { + "epoch": 0.74, + "eval_loss": 3.8828125, + "eval_runtime": 33.0139, + "eval_samples_per_second": 18240.416, + "eval_steps_per_second": 17.841, + "step": 74000 + }, + { + "epoch": 0.741, + "grad_norm": 0.08998531848192215, + "learning_rate": 0.00016428801936338143, + "loss": 3.4437, + "step": 74100 + }, + { + "epoch": 0.742, + "grad_norm": 0.09501896053552628, + "learning_rate": 0.00016417727582577964, + "loss": 3.4427, + "step": 74200 + }, + { + "epoch": 0.743, + "grad_norm": 0.10081542283296585, + "learning_rate": 0.00016406675593723583, + "loss": 3.4416, + "step": 74300 + }, + { + "epoch": 0.744, + "grad_norm": 0.08788462728261948, + "learning_rate": 0.00016395645894598824, + "loss": 3.4433, + "step": 74400 + }, + { + "epoch": 0.745, + "grad_norm": 0.13512863218784332, + "learning_rate": 0.0001638463841038081, + "loss": 3.443, + "step": 74500 + }, + { + "epoch": 0.746, + "grad_norm": 0.08741659671068192, + "learning_rate": 0.00016373653066597825, + "loss": 3.4422, + "step": 74600 + }, + { + "epoch": 0.747, + "grad_norm": 0.09418010711669922, + "learning_rate": 0.000163626897891272, + "loss": 3.4409, + "step": 74700 + }, + { + "epoch": 0.748, + "grad_norm": 0.09971028566360474, + "learning_rate": 0.00016351748504193217, + "loss": 3.4415, + "step": 74800 + }, + { + "epoch": 0.749, + "grad_norm": 0.10280565172433853, + "learning_rate": 0.0001634082913836501, + "loss": 3.4405, + "step": 74900 + }, + { + "epoch": 0.75, + "grad_norm": 0.09686677157878876, + "learning_rate": 0.00016329931618554522, + "loss": 3.4416, + "step": 75000 + }, + { + "epoch": 0.75, + "eval_loss": 3.875, + "eval_runtime": 32.9976, + "eval_samples_per_second": 18249.452, + "eval_steps_per_second": 17.85, + "step": 75000 + }, + { + "epoch": 0.751, + "grad_norm": 0.08808457851409912, + "learning_rate": 0.00016319055872014417, + "loss": 3.4405, + "step": 75100 + }, + { + "epoch": 0.752, + "grad_norm": 0.09505708515644073, + "learning_rate": 0.00016308201826336055, + "loss": 3.441, + "step": 75200 + }, + { + "epoch": 0.753, + "grad_norm": 0.08924631774425507, + "learning_rate": 0.00016297369409447486, + "loss": 3.4413, + "step": 75300 + }, + { + "epoch": 0.754, + "grad_norm": 0.09695728868246078, + "learning_rate": 0.00016286558549611407, + "loss": 3.4417, + "step": 75400 + }, + { + "epoch": 0.755, + "grad_norm": 0.08458244055509567, + "learning_rate": 0.00016275769175423189, + "loss": 3.4414, + "step": 75500 + }, + { + "epoch": 0.756, + "grad_norm": 0.09474821388721466, + "learning_rate": 0.00016265001215808885, + "loss": 3.4387, + "step": 75600 + }, + { + "epoch": 0.757, + "grad_norm": 0.0918809249997139, + "learning_rate": 0.00016254254600023274, + "loss": 3.4394, + "step": 75700 + }, + { + "epoch": 0.758, + "grad_norm": 0.10976070165634155, + "learning_rate": 0.000162435292576479, + "loss": 3.4403, + "step": 75800 + }, + { + "epoch": 0.759, + "grad_norm": 0.1117151528596878, + "learning_rate": 0.00016232825118589134, + "loss": 3.4399, + "step": 75900 + }, + { + "epoch": 0.76, + "grad_norm": 0.12362287938594818, + "learning_rate": 0.00016222142113076255, + "loss": 3.4402, + "step": 76000 + }, + { + "epoch": 0.76, + "eval_loss": 3.8828125, + "eval_runtime": 32.9906, + "eval_samples_per_second": 18253.325, + "eval_steps_per_second": 17.854, + "step": 76000 + }, + { + "epoch": 0.761, + "grad_norm": 0.08545960485935211, + "learning_rate": 0.0001621148017165954, + "loss": 3.4404, + "step": 76100 + }, + { + "epoch": 0.762, + "grad_norm": 0.11282055079936981, + "learning_rate": 0.00016200839225208365, + "loss": 3.4406, + "step": 76200 + }, + { + "epoch": 0.763, + "grad_norm": 0.1176215335726738, + "learning_rate": 0.00016190219204909316, + "loss": 3.4415, + "step": 76300 + }, + { + "epoch": 0.764, + "grad_norm": 0.09560921788215637, + "learning_rate": 0.0001617962004226434, + "loss": 3.4413, + "step": 76400 + }, + { + "epoch": 0.765, + "grad_norm": 0.09105270355939865, + "learning_rate": 0.00016169041669088867, + "loss": 3.4409, + "step": 76500 + }, + { + "epoch": 0.766, + "grad_norm": 0.0954379066824913, + "learning_rate": 0.00016158484017509978, + "loss": 3.4382, + "step": 76600 + }, + { + "epoch": 0.767, + "grad_norm": 0.09138867259025574, + "learning_rate": 0.00016147947019964576, + "loss": 3.4412, + "step": 76700 + }, + { + "epoch": 0.768, + "grad_norm": 0.09806874394416809, + "learning_rate": 0.0001613743060919757, + "loss": 3.442, + "step": 76800 + }, + { + "epoch": 0.769, + "grad_norm": 0.08680954575538635, + "learning_rate": 0.00016126934718260072, + "loss": 3.4406, + "step": 76900 + }, + { + "epoch": 0.77, + "grad_norm": 0.10539354383945465, + "learning_rate": 0.00016116459280507607, + "loss": 3.4412, + "step": 77000 + }, + { + "epoch": 0.77, + "eval_loss": 3.88671875, + "eval_runtime": 33.0031, + "eval_samples_per_second": 18246.399, + "eval_steps_per_second": 17.847, + "step": 77000 + }, + { + "epoch": 0.771, + "grad_norm": 0.09596407413482666, + "learning_rate": 0.0001610600422959833, + "loss": 3.439, + "step": 77100 + }, + { + "epoch": 0.772, + "grad_norm": 0.09768424183130264, + "learning_rate": 0.00016095569499491262, + "loss": 3.4379, + "step": 77200 + }, + { + "epoch": 0.773, + "grad_norm": 0.09115412831306458, + "learning_rate": 0.0001608515502444456, + "loss": 3.4381, + "step": 77300 + }, + { + "epoch": 0.774, + "grad_norm": 0.11213884502649307, + "learning_rate": 0.00016074760739013738, + "loss": 3.4402, + "step": 77400 + }, + { + "epoch": 0.775, + "grad_norm": 0.08757272362709045, + "learning_rate": 0.00016064386578049978, + "loss": 3.438, + "step": 77500 + }, + { + "epoch": 0.776, + "grad_norm": 0.08894165605306625, + "learning_rate": 0.0001605403247669839, + "loss": 3.4417, + "step": 77600 + }, + { + "epoch": 0.777, + "grad_norm": 0.10517994314432144, + "learning_rate": 0.00016043698370396314, + "loss": 3.4397, + "step": 77700 + }, + { + "epoch": 0.778, + "grad_norm": 0.08438605815172195, + "learning_rate": 0.00016033384194871647, + "loss": 3.4374, + "step": 77800 + }, + { + "epoch": 0.779, + "grad_norm": 0.1192629337310791, + "learning_rate": 0.00016023089886141128, + "loss": 3.4374, + "step": 77900 + }, + { + "epoch": 0.78, + "grad_norm": 0.11247923970222473, + "learning_rate": 0.00016012815380508712, + "loss": 3.438, + "step": 78000 + }, + { + "epoch": 0.78, + "eval_loss": 3.875, + "eval_runtime": 32.9875, + "eval_samples_per_second": 18255.029, + "eval_steps_per_second": 17.855, + "step": 78000 + }, + { + "epoch": 0.781, + "grad_norm": 0.1209215372800827, + "learning_rate": 0.0001600256061456389, + "loss": 3.4371, + "step": 78100 + }, + { + "epoch": 0.782, + "grad_norm": 0.09779994189739227, + "learning_rate": 0.00015992325525180032, + "loss": 3.4398, + "step": 78200 + }, + { + "epoch": 0.783, + "grad_norm": 0.09656972438097, + "learning_rate": 0.00015982110049512805, + "loss": 3.4385, + "step": 78300 + }, + { + "epoch": 0.784, + "grad_norm": 0.08916040509939194, + "learning_rate": 0.000159719141249985, + "loss": 3.4366, + "step": 78400 + }, + { + "epoch": 0.785, + "grad_norm": 0.09170043468475342, + "learning_rate": 0.00015961737689352442, + "loss": 3.438, + "step": 78500 + }, + { + "epoch": 0.786, + "grad_norm": 0.09335237741470337, + "learning_rate": 0.0001595158068056741, + "loss": 3.438, + "step": 78600 + }, + { + "epoch": 0.787, + "grad_norm": 0.09777306765317917, + "learning_rate": 0.00015941443036912014, + "loss": 3.4364, + "step": 78700 + }, + { + "epoch": 0.788, + "grad_norm": 0.08855511248111725, + "learning_rate": 0.00015931324696929155, + "loss": 3.4368, + "step": 78800 + }, + { + "epoch": 0.789, + "grad_norm": 0.0980067327618599, + "learning_rate": 0.00015921225599434429, + "loss": 3.4375, + "step": 78900 + }, + { + "epoch": 0.79, + "grad_norm": 0.09301699697971344, + "learning_rate": 0.000159111456835146, + "loss": 3.4359, + "step": 79000 + }, + { + "epoch": 0.79, + "eval_loss": 3.875, + "eval_runtime": 33.0055, + "eval_samples_per_second": 18245.074, + "eval_steps_per_second": 17.846, + "step": 79000 + }, + { + "epoch": 0.791, + "grad_norm": 0.10222512483596802, + "learning_rate": 0.00015901084888526044, + "loss": 3.4359, + "step": 79100 + }, + { + "epoch": 0.792, + "grad_norm": 0.09246222674846649, + "learning_rate": 0.00015891043154093204, + "loss": 3.4358, + "step": 79200 + }, + { + "epoch": 0.793, + "grad_norm": 0.11724498867988586, + "learning_rate": 0.00015881020420107104, + "loss": 3.437, + "step": 79300 + }, + { + "epoch": 0.794, + "grad_norm": 0.08812806755304337, + "learning_rate": 0.0001587101662672379, + "loss": 3.4366, + "step": 79400 + }, + { + "epoch": 0.795, + "grad_norm": 0.11343426257371902, + "learning_rate": 0.00015861031714362882, + "loss": 3.4361, + "step": 79500 + }, + { + "epoch": 0.796, + "grad_norm": 0.08835262805223465, + "learning_rate": 0.00015851065623706038, + "loss": 3.4358, + "step": 79600 + }, + { + "epoch": 0.797, + "grad_norm": 0.08559754490852356, + "learning_rate": 0.00015841118295695486, + "loss": 3.4342, + "step": 79700 + }, + { + "epoch": 0.798, + "grad_norm": 0.13493484258651733, + "learning_rate": 0.0001583118967153259, + "loss": 3.4362, + "step": 79800 + }, + { + "epoch": 0.799, + "grad_norm": 0.08687994629144669, + "learning_rate": 0.00015821279692676326, + "loss": 3.4368, + "step": 79900 + }, + { + "epoch": 0.8, + "grad_norm": 0.08848557621240616, + "learning_rate": 0.00015811388300841897, + "loss": 3.4343, + "step": 80000 + }, + { + "epoch": 0.8, + "eval_loss": 3.869140625, + "eval_runtime": 32.9882, + "eval_samples_per_second": 18254.673, + "eval_steps_per_second": 17.855, + "step": 80000 + }, + { + "epoch": 0.801, + "grad_norm": 0.08878805488348007, + "learning_rate": 0.00015801515437999242, + "loss": 3.4347, + "step": 80100 + }, + { + "epoch": 0.802, + "grad_norm": 0.09379617869853973, + "learning_rate": 0.00015791661046371636, + "loss": 3.4347, + "step": 80200 + }, + { + "epoch": 0.803, + "grad_norm": 0.08302991092205048, + "learning_rate": 0.00015781825068434261, + "loss": 3.4346, + "step": 80300 + }, + { + "epoch": 0.804, + "grad_norm": 0.10080291330814362, + "learning_rate": 0.00015772007446912791, + "loss": 3.4362, + "step": 80400 + }, + { + "epoch": 0.805, + "grad_norm": 0.1289501041173935, + "learning_rate": 0.00015762208124782013, + "loss": 3.4357, + "step": 80500 + }, + { + "epoch": 0.806, + "grad_norm": 0.0920480489730835, + "learning_rate": 0.00015752427045264398, + "loss": 3.4349, + "step": 80600 + }, + { + "epoch": 0.807, + "grad_norm": 0.09406436234712601, + "learning_rate": 0.00015742664151828745, + "loss": 3.4365, + "step": 80700 + }, + { + "epoch": 0.808, + "grad_norm": 0.09713851660490036, + "learning_rate": 0.00015732919388188816, + "loss": 3.4365, + "step": 80800 + }, + { + "epoch": 0.809, + "grad_norm": 0.09240105003118515, + "learning_rate": 0.0001572319269830195, + "loss": 3.4357, + "step": 80900 + }, + { + "epoch": 0.81, + "grad_norm": 0.09189952164888382, + "learning_rate": 0.00015713484026367723, + "loss": 3.4325, + "step": 81000 + }, + { + "epoch": 0.81, + "eval_loss": 3.87109375, + "eval_runtime": 33.0008, + "eval_samples_per_second": 18247.706, + "eval_steps_per_second": 17.848, + "step": 81000 + }, + { + "epoch": 0.811, + "grad_norm": 0.09463001787662506, + "learning_rate": 0.00015703793316826606, + "loss": 3.4342, + "step": 81100 + }, + { + "epoch": 0.812, + "grad_norm": 0.11351007223129272, + "learning_rate": 0.0001569412051435861, + "loss": 3.4349, + "step": 81200 + }, + { + "epoch": 0.813, + "grad_norm": 0.09494343400001526, + "learning_rate": 0.00015684465563882, + "loss": 3.435, + "step": 81300 + }, + { + "epoch": 0.814, + "grad_norm": 0.08718815445899963, + "learning_rate": 0.00015674828410551923, + "loss": 3.4344, + "step": 81400 + }, + { + "epoch": 0.815, + "grad_norm": 0.08435869216918945, + "learning_rate": 0.00015665208999759148, + "loss": 3.4326, + "step": 81500 + }, + { + "epoch": 0.816, + "grad_norm": 0.11424301564693451, + "learning_rate": 0.0001565560727712874, + "loss": 3.4322, + "step": 81600 + }, + { + "epoch": 0.817, + "grad_norm": 0.11047474294900894, + "learning_rate": 0.0001564602318851877, + "loss": 3.4343, + "step": 81700 + }, + { + "epoch": 0.818, + "grad_norm": 0.09242834150791168, + "learning_rate": 0.00015636456680019053, + "loss": 3.4329, + "step": 81800 + }, + { + "epoch": 0.819, + "grad_norm": 0.10262425988912582, + "learning_rate": 0.00015626907697949846, + "loss": 3.4338, + "step": 81900 + }, + { + "epoch": 0.82, + "grad_norm": 0.0893552154302597, + "learning_rate": 0.00015617376188860606, + "loss": 3.4346, + "step": 82000 + }, + { + "epoch": 0.82, + "eval_loss": 3.876953125, + "eval_runtime": 32.9906, + "eval_samples_per_second": 18253.339, + "eval_steps_per_second": 17.854, + "step": 82000 + }, + { + "epoch": 0.821, + "grad_norm": 0.0904051810503006, + "learning_rate": 0.00015607862099528718, + "loss": 3.433, + "step": 82100 + }, + { + "epoch": 0.822, + "grad_norm": 0.08803316950798035, + "learning_rate": 0.00015598365376958255, + "loss": 3.431, + "step": 82200 + }, + { + "epoch": 0.823, + "grad_norm": 0.08486661314964294, + "learning_rate": 0.00015588885968378734, + "loss": 3.4327, + "step": 82300 + }, + { + "epoch": 0.824, + "grad_norm": 0.1493000090122223, + "learning_rate": 0.00015579423821243896, + "loss": 3.4336, + "step": 82400 + }, + { + "epoch": 0.825, + "grad_norm": 0.08111909031867981, + "learning_rate": 0.0001556997888323046, + "loss": 3.4309, + "step": 82500 + }, + { + "epoch": 0.826, + "grad_norm": 0.15636824071407318, + "learning_rate": 0.0001556055110223693, + "loss": 3.433, + "step": 82600 + }, + { + "epoch": 0.827, + "grad_norm": 0.09289142489433289, + "learning_rate": 0.00015551140426382372, + "loss": 3.4333, + "step": 82700 + }, + { + "epoch": 0.828, + "grad_norm": 0.10629736632108688, + "learning_rate": 0.0001554174680400523, + "loss": 3.432, + "step": 82800 + }, + { + "epoch": 0.829, + "grad_norm": 0.11100473254919052, + "learning_rate": 0.0001553237018366212, + "loss": 3.433, + "step": 82900 + }, + { + "epoch": 0.83, + "grad_norm": 0.09321753680706024, + "learning_rate": 0.00015523010514126656, + "loss": 3.4343, + "step": 83000 + }, + { + "epoch": 0.83, + "eval_loss": 3.87109375, + "eval_runtime": 32.9766, + "eval_samples_per_second": 18261.078, + "eval_steps_per_second": 17.861, + "step": 83000 + }, + { + "epoch": 0.831, + "grad_norm": 0.1051269918680191, + "learning_rate": 0.00015513667744388273, + "loss": 3.4318, + "step": 83100 + }, + { + "epoch": 0.832, + "grad_norm": 0.108842633664608, + "learning_rate": 0.00015504341823651055, + "loss": 3.4324, + "step": 83200 + }, + { + "epoch": 0.833, + "grad_norm": 0.11733662337064743, + "learning_rate": 0.00015495032701332584, + "loss": 3.434, + "step": 83300 + }, + { + "epoch": 0.834, + "grad_norm": 0.12677276134490967, + "learning_rate": 0.00015485740327062772, + "loss": 3.4328, + "step": 83400 + }, + { + "epoch": 0.835, + "grad_norm": 0.08611226826906204, + "learning_rate": 0.00015476464650682737, + "loss": 3.4333, + "step": 83500 + }, + { + "epoch": 0.836, + "grad_norm": 0.08939053118228912, + "learning_rate": 0.0001546720562224365, + "loss": 3.4316, + "step": 83600 + }, + { + "epoch": 0.837, + "grad_norm": 0.10289647430181503, + "learning_rate": 0.0001545796319200561, + "loss": 3.4311, + "step": 83700 + }, + { + "epoch": 0.838, + "grad_norm": 0.0950452983379364, + "learning_rate": 0.00015448737310436522, + "loss": 3.4332, + "step": 83800 + }, + { + "epoch": 0.839, + "grad_norm": 0.09242187440395355, + "learning_rate": 0.00015439527928210988, + "loss": 3.4311, + "step": 83900 + }, + { + "epoch": 0.84, + "grad_norm": 0.10447569936513901, + "learning_rate": 0.00015430334996209192, + "loss": 3.4316, + "step": 84000 + }, + { + "epoch": 0.84, + "eval_loss": 3.875, + "eval_runtime": 32.9725, + "eval_samples_per_second": 18263.359, + "eval_steps_per_second": 17.863, + "step": 84000 + }, + { + "epoch": 0.841, + "grad_norm": 0.09993547201156616, + "learning_rate": 0.0001542115846551579, + "loss": 3.4326, + "step": 84100 + }, + { + "epoch": 0.842, + "grad_norm": 0.0941188782453537, + "learning_rate": 0.00015411998287418846, + "loss": 3.4318, + "step": 84200 + }, + { + "epoch": 0.843, + "grad_norm": 0.12041068077087402, + "learning_rate": 0.00015402854413408716, + "loss": 3.4319, + "step": 84300 + }, + { + "epoch": 0.844, + "grad_norm": 0.09113665670156479, + "learning_rate": 0.00015393726795176978, + "loss": 3.4323, + "step": 84400 + }, + { + "epoch": 0.845, + "grad_norm": 0.11387240141630173, + "learning_rate": 0.00015384615384615385, + "loss": 3.4301, + "step": 84500 + }, + { + "epoch": 0.846, + "grad_norm": 0.09158744663000107, + "learning_rate": 0.0001537552013381475, + "loss": 3.4298, + "step": 84600 + }, + { + "epoch": 0.847, + "grad_norm": 0.08281726390123367, + "learning_rate": 0.00015366440995063938, + "loss": 3.4319, + "step": 84700 + }, + { + "epoch": 0.848, + "grad_norm": 0.08520444482564926, + "learning_rate": 0.0001535737792084878, + "loss": 3.4314, + "step": 84800 + }, + { + "epoch": 0.849, + "grad_norm": 0.08099710196256638, + "learning_rate": 0.0001534833086385105, + "loss": 3.4322, + "step": 84900 + }, + { + "epoch": 0.85, + "grad_norm": 0.08628775179386139, + "learning_rate": 0.00015339299776947408, + "loss": 3.4306, + "step": 85000 + }, + { + "epoch": 0.85, + "eval_loss": 3.873046875, + "eval_runtime": 32.9817, + "eval_samples_per_second": 18258.237, + "eval_steps_per_second": 17.858, + "step": 85000 + }, + { + "epoch": 0.851, + "grad_norm": 0.08802857249975204, + "learning_rate": 0.00015330284613208398, + "loss": 3.4298, + "step": 85100 + }, + { + "epoch": 0.852, + "grad_norm": 0.09676956385374069, + "learning_rate": 0.00015321285325897388, + "loss": 3.4307, + "step": 85200 + }, + { + "epoch": 0.853, + "grad_norm": 0.10504265874624252, + "learning_rate": 0.00015312301868469587, + "loss": 3.4301, + "step": 85300 + }, + { + "epoch": 0.854, + "grad_norm": 0.08877792209386826, + "learning_rate": 0.00015303334194570998, + "loss": 3.4306, + "step": 85400 + }, + { + "epoch": 0.855, + "grad_norm": 0.12948188185691833, + "learning_rate": 0.0001529438225803745, + "loss": 3.4299, + "step": 85500 + }, + { + "epoch": 0.856, + "grad_norm": 0.08437684923410416, + "learning_rate": 0.00015285446012893578, + "loss": 3.4306, + "step": 85600 + }, + { + "epoch": 0.857, + "grad_norm": 0.08995580673217773, + "learning_rate": 0.00015276525413351826, + "loss": 3.4316, + "step": 85700 + }, + { + "epoch": 0.858, + "grad_norm": 0.08715008944272995, + "learning_rate": 0.00015267620413811483, + "loss": 3.4298, + "step": 85800 + }, + { + "epoch": 0.859, + "grad_norm": 0.0873521938920021, + "learning_rate": 0.0001525873096885769, + "loss": 3.428, + "step": 85900 + }, + { + "epoch": 0.86, + "grad_norm": 0.08357945084571838, + "learning_rate": 0.00015249857033260467, + "loss": 3.4303, + "step": 86000 + }, + { + "epoch": 0.86, + "eval_loss": 3.873046875, + "eval_runtime": 32.9692, + "eval_samples_per_second": 18265.15, + "eval_steps_per_second": 17.865, + "step": 86000 + }, + { + "epoch": 0.861, + "grad_norm": 0.0914229154586792, + "learning_rate": 0.00015240998561973752, + "loss": 3.4303, + "step": 86100 + }, + { + "epoch": 0.862, + "grad_norm": 0.10973150283098221, + "learning_rate": 0.00015232155510134433, + "loss": 3.4277, + "step": 86200 + }, + { + "epoch": 0.863, + "grad_norm": 0.1186969205737114, + "learning_rate": 0.00015223327833061405, + "loss": 3.4279, + "step": 86300 + }, + { + "epoch": 0.864, + "grad_norm": 0.08916877210140228, + "learning_rate": 0.00015214515486254611, + "loss": 3.4301, + "step": 86400 + }, + { + "epoch": 0.865, + "grad_norm": 0.09595561027526855, + "learning_rate": 0.0001520571842539411, + "loss": 3.4272, + "step": 86500 + }, + { + "epoch": 0.866, + "grad_norm": 0.10776854306459427, + "learning_rate": 0.00015196936606339122, + "loss": 3.4285, + "step": 86600 + }, + { + "epoch": 0.867, + "grad_norm": 0.09018661081790924, + "learning_rate": 0.00015188169985127126, + "loss": 3.4284, + "step": 86700 + }, + { + "epoch": 0.868, + "grad_norm": 0.09676007181406021, + "learning_rate": 0.0001517941851797291, + "loss": 3.428, + "step": 86800 + }, + { + "epoch": 0.869, + "grad_norm": 0.09606127440929413, + "learning_rate": 0.0001517068216126766, + "loss": 3.4301, + "step": 86900 + }, + { + "epoch": 0.87, + "grad_norm": 0.0857662633061409, + "learning_rate": 0.0001516196087157807, + "loss": 3.431, + "step": 87000 + }, + { + "epoch": 0.87, + "eval_loss": 3.87109375, + "eval_runtime": 32.9775, + "eval_samples_per_second": 18260.568, + "eval_steps_per_second": 17.861, + "step": 87000 + }, + { + "epoch": 0.871, + "grad_norm": 0.11101428419351578, + "learning_rate": 0.0001515325460564539, + "loss": 3.4315, + "step": 87100 + }, + { + "epoch": 0.872, + "grad_norm": 0.10998611152172089, + "learning_rate": 0.00015144563320384566, + "loss": 3.4312, + "step": 87200 + }, + { + "epoch": 0.873, + "grad_norm": 0.0887598916888237, + "learning_rate": 0.0001513588697288333, + "loss": 3.4287, + "step": 87300 + }, + { + "epoch": 0.874, + "grad_norm": 0.09869083762168884, + "learning_rate": 0.00015127225520401288, + "loss": 3.4286, + "step": 87400 + }, + { + "epoch": 0.875, + "grad_norm": 0.09993433952331543, + "learning_rate": 0.0001511857892036909, + "loss": 3.4268, + "step": 87500 + }, + { + "epoch": 0.876, + "grad_norm": 0.09299583733081818, + "learning_rate": 0.00015109947130387486, + "loss": 3.4288, + "step": 87600 + }, + { + "epoch": 0.877, + "grad_norm": 0.09471836686134338, + "learning_rate": 0.00015101330108226502, + "loss": 3.4278, + "step": 87700 + }, + { + "epoch": 0.878, + "grad_norm": 0.10189653187990189, + "learning_rate": 0.00015092727811824553, + "loss": 3.4277, + "step": 87800 + }, + { + "epoch": 0.879, + "grad_norm": 0.09086822718381882, + "learning_rate": 0.00015084140199287576, + "loss": 3.4283, + "step": 87900 + }, + { + "epoch": 0.88, + "grad_norm": 0.08800884336233139, + "learning_rate": 0.00015075567228888182, + "loss": 3.43, + "step": 88000 + }, + { + "epoch": 0.88, + "eval_loss": 3.8671875, + "eval_runtime": 32.9826, + "eval_samples_per_second": 18257.767, + "eval_steps_per_second": 17.858, + "step": 88000 + }, + { + "epoch": 0.881, + "grad_norm": 0.08497955650091171, + "learning_rate": 0.00015067008859064807, + "loss": 3.4282, + "step": 88100 + }, + { + "epoch": 0.882, + "grad_norm": 0.09592936187982559, + "learning_rate": 0.00015058465048420854, + "loss": 3.4292, + "step": 88200 + }, + { + "epoch": 0.883, + "grad_norm": 0.08380814641714096, + "learning_rate": 0.00015049935755723866, + "loss": 3.4281, + "step": 88300 + }, + { + "epoch": 0.884, + "grad_norm": 0.08948108553886414, + "learning_rate": 0.0001504142093990467, + "loss": 3.4273, + "step": 88400 + }, + { + "epoch": 0.885, + "grad_norm": 0.09598647058010101, + "learning_rate": 0.00015032920560056578, + "loss": 3.4282, + "step": 88500 + }, + { + "epoch": 0.886, + "grad_norm": 0.09412466734647751, + "learning_rate": 0.00015024434575434525, + "loss": 3.4272, + "step": 88600 + }, + { + "epoch": 0.887, + "grad_norm": 0.08592375367879868, + "learning_rate": 0.0001501596294545428, + "loss": 3.4258, + "step": 88700 + }, + { + "epoch": 0.888, + "grad_norm": 0.09520655870437622, + "learning_rate": 0.00015007505629691605, + "loss": 3.4295, + "step": 88800 + }, + { + "epoch": 0.889, + "grad_norm": 0.08593907207250595, + "learning_rate": 0.0001499906258788147, + "loss": 3.4263, + "step": 88900 + }, + { + "epoch": 0.89, + "grad_norm": 0.09926886111497879, + "learning_rate": 0.00014990633779917229, + "loss": 3.4279, + "step": 89000 + }, + { + "epoch": 0.89, + "eval_loss": 3.86328125, + "eval_runtime": 32.9884, + "eval_samples_per_second": 18254.532, + "eval_steps_per_second": 17.855, + "step": 89000 + }, + { + "epoch": 0.891, + "grad_norm": 0.09873060137033463, + "learning_rate": 0.00014982219165849827, + "loss": 3.4272, + "step": 89100 + }, + { + "epoch": 0.892, + "grad_norm": 0.09635820984840393, + "learning_rate": 0.00014973818705886998, + "loss": 3.4278, + "step": 89200 + }, + { + "epoch": 0.893, + "grad_norm": 0.08815903961658478, + "learning_rate": 0.00014965432360392487, + "loss": 3.4278, + "step": 89300 + }, + { + "epoch": 0.894, + "grad_norm": 0.08887197822332382, + "learning_rate": 0.00014957060089885261, + "loss": 3.4264, + "step": 89400 + }, + { + "epoch": 0.895, + "grad_norm": 0.08929969370365143, + "learning_rate": 0.00014948701855038718, + "loss": 3.4282, + "step": 89500 + }, + { + "epoch": 0.896, + "grad_norm": 0.09338970482349396, + "learning_rate": 0.00014940357616679919, + "loss": 3.4288, + "step": 89600 + }, + { + "epoch": 0.897, + "grad_norm": 0.1054818257689476, + "learning_rate": 0.00014932027335788826, + "loss": 3.4285, + "step": 89700 + }, + { + "epoch": 0.898, + "grad_norm": 0.09273470193147659, + "learning_rate": 0.0001492371097349751, + "loss": 3.4276, + "step": 89800 + }, + { + "epoch": 0.899, + "grad_norm": 0.08965855836868286, + "learning_rate": 0.00014915408491089415, + "loss": 3.4274, + "step": 89900 + }, + { + "epoch": 0.9, + "grad_norm": 0.11126144975423813, + "learning_rate": 0.00014907119849998598, + "loss": 3.4265, + "step": 90000 + }, + { + "epoch": 0.9, + "eval_loss": 3.87109375, + "eval_runtime": 32.9878, + "eval_samples_per_second": 18254.881, + "eval_steps_per_second": 17.855, + "step": 90000 + }, + { + "epoch": 0.901, + "grad_norm": 0.09338975697755814, + "learning_rate": 0.00014898845011808955, + "loss": 3.4256, + "step": 90100 + }, + { + "epoch": 0.902, + "grad_norm": 0.09844280779361725, + "learning_rate": 0.00014890583938253493, + "loss": 3.4274, + "step": 90200 + }, + { + "epoch": 0.903, + "grad_norm": 0.08697222173213959, + "learning_rate": 0.0001488233659121359, + "loss": 3.4254, + "step": 90300 + }, + { + "epoch": 0.904, + "grad_norm": 0.09345566481351852, + "learning_rate": 0.0001487410293271824, + "loss": 3.4268, + "step": 90400 + }, + { + "epoch": 0.905, + "grad_norm": 0.08729152381420135, + "learning_rate": 0.00014865882924943328, + "loss": 3.4251, + "step": 90500 + }, + { + "epoch": 0.906, + "grad_norm": 0.12593166530132294, + "learning_rate": 0.00014857676530210894, + "loss": 3.4275, + "step": 90600 + }, + { + "epoch": 0.907, + "grad_norm": 0.0890730693936348, + "learning_rate": 0.00014849483710988428, + "loss": 3.4251, + "step": 90700 + }, + { + "epoch": 0.908, + "grad_norm": 0.08751527220010757, + "learning_rate": 0.0001484130442988812, + "loss": 3.4244, + "step": 90800 + }, + { + "epoch": 0.909, + "grad_norm": 0.11215963959693909, + "learning_rate": 0.00014833138649666158, + "loss": 3.424, + "step": 90900 + }, + { + "epoch": 0.91, + "grad_norm": 0.101831816136837, + "learning_rate": 0.00014824986333222023, + "loss": 3.4266, + "step": 91000 + }, + { + "epoch": 0.91, + "eval_loss": 3.857421875, + "eval_runtime": 32.9906, + "eval_samples_per_second": 18253.323, + "eval_steps_per_second": 17.854, + "step": 91000 + }, + { + "epoch": 0.911, + "grad_norm": 0.09032022207975388, + "learning_rate": 0.00014816847443597765, + "loss": 3.4258, + "step": 91100 + }, + { + "epoch": 0.912, + "grad_norm": 0.11087517440319061, + "learning_rate": 0.00014808721943977308, + "loss": 3.4257, + "step": 91200 + }, + { + "epoch": 0.913, + "grad_norm": 0.08949613571166992, + "learning_rate": 0.00014800609797685756, + "loss": 3.4256, + "step": 91300 + }, + { + "epoch": 0.914, + "grad_norm": 0.08762556314468384, + "learning_rate": 0.00014792510968188681, + "loss": 3.4229, + "step": 91400 + }, + { + "epoch": 0.915, + "grad_norm": 0.09100620448589325, + "learning_rate": 0.00014784425419091458, + "loss": 3.4263, + "step": 91500 + }, + { + "epoch": 0.916, + "grad_norm": 0.10705144703388214, + "learning_rate": 0.00014776353114138543, + "loss": 3.4235, + "step": 91600 + }, + { + "epoch": 0.917, + "grad_norm": 0.1440095603466034, + "learning_rate": 0.00014768294017212822, + "loss": 3.4259, + "step": 91700 + }, + { + "epoch": 0.918, + "grad_norm": 0.11515694111585617, + "learning_rate": 0.00014760248092334923, + "loss": 3.4233, + "step": 91800 + }, + { + "epoch": 0.919, + "grad_norm": 0.08891148865222931, + "learning_rate": 0.00014752215303662526, + "loss": 3.4228, + "step": 91900 + }, + { + "epoch": 0.92, + "grad_norm": 0.0852489098906517, + "learning_rate": 0.00014744195615489715, + "loss": 3.4253, + "step": 92000 + }, + { + "epoch": 0.92, + "eval_loss": 3.869140625, + "eval_runtime": 32.971, + "eval_samples_per_second": 18264.179, + "eval_steps_per_second": 17.864, + "step": 92000 + }, + { + "epoch": 0.921, + "grad_norm": 0.09289970248937607, + "learning_rate": 0.00014736188992246294, + "loss": 3.4251, + "step": 92100 + }, + { + "epoch": 0.922, + "grad_norm": 0.09001227468252182, + "learning_rate": 0.0001472819539849714, + "loss": 3.4238, + "step": 92200 + }, + { + "epoch": 0.923, + "grad_norm": 0.08918751776218414, + "learning_rate": 0.0001472021479894153, + "loss": 3.4254, + "step": 92300 + }, + { + "epoch": 0.924, + "grad_norm": 0.11080797016620636, + "learning_rate": 0.00014712247158412493, + "loss": 3.4238, + "step": 92400 + }, + { + "epoch": 0.925, + "grad_norm": 0.08473057299852371, + "learning_rate": 0.00014704292441876154, + "loss": 3.426, + "step": 92500 + }, + { + "epoch": 0.926, + "grad_norm": 0.11121894419193268, + "learning_rate": 0.00014696350614431104, + "loss": 3.4233, + "step": 92600 + }, + { + "epoch": 0.927, + "grad_norm": 0.0871448889374733, + "learning_rate": 0.00014688421641307725, + "loss": 3.4233, + "step": 92700 + }, + { + "epoch": 0.928, + "grad_norm": 0.0890325978398323, + "learning_rate": 0.00014680505487867586, + "loss": 3.4248, + "step": 92800 + }, + { + "epoch": 0.929, + "grad_norm": 0.10418891906738281, + "learning_rate": 0.0001467260211960279, + "loss": 3.4221, + "step": 92900 + }, + { + "epoch": 0.93, + "grad_norm": 0.0831473171710968, + "learning_rate": 0.0001466471150213533, + "loss": 3.4226, + "step": 93000 + }, + { + "epoch": 0.93, + "eval_loss": 3.869140625, + "eval_runtime": 32.9976, + "eval_samples_per_second": 18249.449, + "eval_steps_per_second": 17.85, + "step": 93000 + }, + { + "epoch": 0.931, + "grad_norm": 0.0834508016705513, + "learning_rate": 0.00014656833601216487, + "loss": 3.4233, + "step": 93100 + }, + { + "epoch": 0.932, + "grad_norm": 0.0903562530875206, + "learning_rate": 0.0001464896838272619, + "loss": 3.424, + "step": 93200 + }, + { + "epoch": 0.933, + "grad_norm": 0.1055062785744667, + "learning_rate": 0.00014641115812672397, + "loss": 3.4246, + "step": 93300 + }, + { + "epoch": 0.934, + "grad_norm": 0.08662033081054688, + "learning_rate": 0.00014633275857190483, + "loss": 3.4235, + "step": 93400 + }, + { + "epoch": 0.935, + "grad_norm": 0.08579885214567184, + "learning_rate": 0.00014625448482542613, + "loss": 3.425, + "step": 93500 + }, + { + "epoch": 0.936, + "grad_norm": 0.09081684798002243, + "learning_rate": 0.00014617633655117154, + "loss": 3.4235, + "step": 93600 + }, + { + "epoch": 0.937, + "grad_norm": 0.11446008086204529, + "learning_rate": 0.00014609831341428047, + "loss": 3.4207, + "step": 93700 + }, + { + "epoch": 0.938, + "grad_norm": 0.09146326035261154, + "learning_rate": 0.00014602041508114227, + "loss": 3.4223, + "step": 93800 + }, + { + "epoch": 0.939, + "grad_norm": 0.09234822541475296, + "learning_rate": 0.00014594264121938996, + "loss": 3.4204, + "step": 93900 + }, + { + "epoch": 0.94, + "grad_norm": 0.09005927294492722, + "learning_rate": 0.00014586499149789455, + "loss": 3.4208, + "step": 94000 + }, + { + "epoch": 0.94, + "eval_loss": 3.859375, + "eval_runtime": 33.0129, + "eval_samples_per_second": 18240.999, + "eval_steps_per_second": 17.842, + "step": 94000 + }, + { + "epoch": 0.941, + "grad_norm": 0.09933404624462128, + "learning_rate": 0.00014578746558675893, + "loss": 3.4215, + "step": 94100 + }, + { + "epoch": 0.942, + "grad_norm": 0.12064062803983688, + "learning_rate": 0.000145710063157312, + "loss": 3.4215, + "step": 94200 + }, + { + "epoch": 0.943, + "grad_norm": 0.09077318012714386, + "learning_rate": 0.00014563278388210303, + "loss": 3.4224, + "step": 94300 + }, + { + "epoch": 0.944, + "grad_norm": 0.10471939295530319, + "learning_rate": 0.0001455556274348955, + "loss": 3.4217, + "step": 94400 + }, + { + "epoch": 0.945, + "grad_norm": 0.11697626858949661, + "learning_rate": 0.00014547859349066158, + "loss": 3.4203, + "step": 94500 + }, + { + "epoch": 0.946, + "grad_norm": 0.10920999944210052, + "learning_rate": 0.00014540168172557633, + "loss": 3.4236, + "step": 94600 + }, + { + "epoch": 0.947, + "grad_norm": 0.10700856149196625, + "learning_rate": 0.0001453248918170118, + "loss": 3.4234, + "step": 94700 + }, + { + "epoch": 0.948, + "grad_norm": 0.0889246016740799, + "learning_rate": 0.0001452482234435317, + "loss": 3.4204, + "step": 94800 + }, + { + "epoch": 0.949, + "grad_norm": 0.09357952326536179, + "learning_rate": 0.00014517167628488533, + "loss": 3.4208, + "step": 94900 + }, + { + "epoch": 0.95, + "grad_norm": 0.09217779338359833, + "learning_rate": 0.00014509525002200232, + "loss": 3.4228, + "step": 95000 + }, + { + "epoch": 0.95, + "eval_loss": 3.865234375, + "eval_runtime": 32.9993, + "eval_samples_per_second": 18248.486, + "eval_steps_per_second": 17.849, + "step": 95000 + }, + { + "epoch": 0.951, + "grad_norm": 0.08662614971399307, + "learning_rate": 0.00014501894433698685, + "loss": 3.4219, + "step": 95100 + }, + { + "epoch": 0.952, + "grad_norm": 0.10022010654211044, + "learning_rate": 0.00014494275891311213, + "loss": 3.4205, + "step": 95200 + }, + { + "epoch": 0.953, + "grad_norm": 0.09261790663003922, + "learning_rate": 0.00014486669343481484, + "loss": 3.4234, + "step": 95300 + }, + { + "epoch": 0.954, + "grad_norm": 0.0873044952750206, + "learning_rate": 0.00014479074758768978, + "loss": 3.4213, + "step": 95400 + }, + { + "epoch": 0.955, + "grad_norm": 0.09158512949943542, + "learning_rate": 0.00014471492105848434, + "loss": 3.4232, + "step": 95500 + }, + { + "epoch": 0.956, + "grad_norm": 0.1242978498339653, + "learning_rate": 0.00014463921353509294, + "loss": 3.4225, + "step": 95600 + }, + { + "epoch": 0.957, + "grad_norm": 0.09228156507015228, + "learning_rate": 0.00014456362470655182, + "loss": 3.4216, + "step": 95700 + }, + { + "epoch": 0.958, + "grad_norm": 0.11788316071033478, + "learning_rate": 0.00014448815426303362, + "loss": 3.4207, + "step": 95800 + }, + { + "epoch": 0.959, + "grad_norm": 0.08355174213647842, + "learning_rate": 0.00014441280189584203, + "loss": 3.4211, + "step": 95900 + }, + { + "epoch": 0.96, + "grad_norm": 0.1150907650589943, + "learning_rate": 0.00014433756729740648, + "loss": 3.421, + "step": 96000 + }, + { + "epoch": 0.96, + "eval_loss": 3.86328125, + "eval_runtime": 32.9989, + "eval_samples_per_second": 18248.745, + "eval_steps_per_second": 17.849, + "step": 96000 + }, + { + "epoch": 0.961, + "grad_norm": 0.09715475887060165, + "learning_rate": 0.00014426245016127675, + "loss": 3.421, + "step": 96100 + }, + { + "epoch": 0.962, + "grad_norm": 0.08516929298639297, + "learning_rate": 0.00014418745018211809, + "loss": 3.4204, + "step": 96200 + }, + { + "epoch": 0.963, + "grad_norm": 0.09556137025356293, + "learning_rate": 0.00014411256705570565, + "loss": 3.4213, + "step": 96300 + }, + { + "epoch": 0.964, + "grad_norm": 0.11192219704389572, + "learning_rate": 0.00014403780047891933, + "loss": 3.42, + "step": 96400 + }, + { + "epoch": 0.965, + "grad_norm": 0.12303122878074646, + "learning_rate": 0.0001439631501497389, + "loss": 3.4192, + "step": 96500 + }, + { + "epoch": 0.966, + "grad_norm": 0.08821086585521698, + "learning_rate": 0.00014388861576723855, + "loss": 3.4192, + "step": 96600 + }, + { + "epoch": 0.967, + "grad_norm": 0.10734008997678757, + "learning_rate": 0.00014381419703158197, + "loss": 3.4211, + "step": 96700 + }, + { + "epoch": 0.968, + "grad_norm": 0.08126262575387955, + "learning_rate": 0.00014373989364401728, + "loss": 3.4206, + "step": 96800 + }, + { + "epoch": 0.969, + "grad_norm": 0.10735704749822617, + "learning_rate": 0.00014366570530687184, + "loss": 3.4211, + "step": 96900 + }, + { + "epoch": 0.97, + "grad_norm": 0.08785036206245422, + "learning_rate": 0.00014359163172354765, + "loss": 3.4198, + "step": 97000 + }, + { + "epoch": 0.97, + "eval_loss": 3.861328125, + "eval_runtime": 32.9832, + "eval_samples_per_second": 18257.433, + "eval_steps_per_second": 17.858, + "step": 97000 + }, + { + "epoch": 0.971, + "grad_norm": 0.0964650809764862, + "learning_rate": 0.00014351767259851572, + "loss": 3.42, + "step": 97100 + }, + { + "epoch": 0.972, + "grad_norm": 0.09006300568580627, + "learning_rate": 0.00014344382763731173, + "loss": 3.4206, + "step": 97200 + }, + { + "epoch": 0.973, + "grad_norm": 0.11106297373771667, + "learning_rate": 0.0001433700965465308, + "loss": 3.4203, + "step": 97300 + }, + { + "epoch": 0.974, + "grad_norm": 0.08400940150022507, + "learning_rate": 0.0001432964790338226, + "loss": 3.421, + "step": 97400 + }, + { + "epoch": 0.975, + "grad_norm": 0.08784104138612747, + "learning_rate": 0.0001432229748078866, + "loss": 3.4188, + "step": 97500 + }, + { + "epoch": 0.976, + "grad_norm": 0.08701920509338379, + "learning_rate": 0.00014314958357846704, + "loss": 3.422, + "step": 97600 + }, + { + "epoch": 0.977, + "grad_norm": 0.09414151310920715, + "learning_rate": 0.00014307630505634844, + "loss": 3.4204, + "step": 97700 + }, + { + "epoch": 0.978, + "grad_norm": 0.10199437290430069, + "learning_rate": 0.00014300313895335043, + "loss": 3.4202, + "step": 97800 + }, + { + "epoch": 0.979, + "grad_norm": 0.09039287269115448, + "learning_rate": 0.00014293008498232322, + "loss": 3.4193, + "step": 97900 + }, + { + "epoch": 0.98, + "grad_norm": 0.10374122112989426, + "learning_rate": 0.00014285714285714284, + "loss": 3.4202, + "step": 98000 + }, + { + "epoch": 0.98, + "eval_loss": 3.859375, + "eval_runtime": 32.984, + "eval_samples_per_second": 18256.958, + "eval_steps_per_second": 17.857, + "step": 98000 + }, + { + "epoch": 0.981, + "grad_norm": 0.0891217291355133, + "learning_rate": 0.00014278431229270645, + "loss": 3.4182, + "step": 98100 + }, + { + "epoch": 0.982, + "grad_norm": 0.0891917422413826, + "learning_rate": 0.0001427115930049275, + "loss": 3.4195, + "step": 98200 + }, + { + "epoch": 0.983, + "grad_norm": 0.11202235519886017, + "learning_rate": 0.0001426389847107313, + "loss": 3.4198, + "step": 98300 + }, + { + "epoch": 0.984, + "grad_norm": 0.11552020907402039, + "learning_rate": 0.00014256648712805025, + "loss": 3.4204, + "step": 98400 + }, + { + "epoch": 0.985, + "grad_norm": 0.09355608373880386, + "learning_rate": 0.0001424940999758193, + "loss": 3.4167, + "step": 98500 + }, + { + "epoch": 0.986, + "grad_norm": 0.10058041661977768, + "learning_rate": 0.00014242182297397128, + "loss": 3.4191, + "step": 98600 + }, + { + "epoch": 0.987, + "grad_norm": 0.09189483523368835, + "learning_rate": 0.0001423496558434325, + "loss": 3.4181, + "step": 98700 + }, + { + "epoch": 0.988, + "grad_norm": 0.12277176976203918, + "learning_rate": 0.00014227759830611806, + "loss": 3.4168, + "step": 98800 + }, + { + "epoch": 0.989, + "grad_norm": 0.09363926202058792, + "learning_rate": 0.0001422056500849275, + "loss": 3.4179, + "step": 98900 + }, + { + "epoch": 0.99, + "grad_norm": 0.08723656833171844, + "learning_rate": 0.0001421338109037403, + "loss": 3.4195, + "step": 99000 + }, + { + "epoch": 0.99, + "eval_loss": 3.861328125, + "eval_runtime": 32.9644, + "eval_samples_per_second": 18267.85, + "eval_steps_per_second": 17.868, + "step": 99000 + }, + { + "epoch": 0.991, + "grad_norm": 0.08475708216428757, + "learning_rate": 0.00014206208048741123, + "loss": 3.4185, + "step": 99100 + }, + { + "epoch": 0.992, + "grad_norm": 0.1032075509428978, + "learning_rate": 0.0001419904585617662, + "loss": 3.4171, + "step": 99200 + }, + { + "epoch": 0.993, + "grad_norm": 0.08446966111660004, + "learning_rate": 0.00014191894485359772, + "loss": 3.419, + "step": 99300 + }, + { + "epoch": 0.994, + "grad_norm": 0.08278857171535492, + "learning_rate": 0.0001418475390906605, + "loss": 3.4181, + "step": 99400 + }, + { + "epoch": 0.995, + "grad_norm": 0.10606228560209274, + "learning_rate": 0.00014177624100166717, + "loss": 3.4171, + "step": 99500 + }, + { + "epoch": 0.996, + "grad_norm": 0.09982995688915253, + "learning_rate": 0.00014170505031628393, + "loss": 3.4193, + "step": 99600 + }, + { + "epoch": 0.997, + "grad_norm": 0.08789572864770889, + "learning_rate": 0.0001416339667651262, + "loss": 3.4177, + "step": 99700 + }, + { + "epoch": 0.998, + "grad_norm": 0.08526172488927841, + "learning_rate": 0.0001415629900797544, + "loss": 3.4187, + "step": 99800 + }, + { + "epoch": 0.999, + "grad_norm": 0.13719169795513153, + "learning_rate": 0.00014149211999266962, + "loss": 3.4189, + "step": 99900 + }, + { + "epoch": 1.0, + "grad_norm": 0.08984668552875519, + "learning_rate": 0.0001414213562373095, + "loss": 3.4164, + "step": 100000 + }, + { + "epoch": 1.0, + "eval_loss": 3.859375, + "eval_runtime": 32.9957, + "eval_samples_per_second": 18250.5, + "eval_steps_per_second": 17.851, + "step": 100000 + }, + { + "epoch": 1.0, + "step": 100000, + "total_flos": 8.175126367943472e+19, + "train_loss": 3.5368977130126953, + "train_runtime": 127327.336, + "train_samples_per_second": 3920.604, + "train_steps_per_second": 0.785 + } + ], + "logging_steps": 100, + "max_steps": 100000, + "num_input_tokens_seen": 0, + "num_train_epochs": 9223372036854775807, + "save_steps": 1000, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 8.175126367943472e+19, + "train_batch_size": 48, + "trial_name": null, + "trial_params": null +}