| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.24, |
| "eval_steps": 1000, |
| "global_step": 12000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": false, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 22.030515670776367, |
| "learning_rate": 7.920000000000001e-05, |
| "loss": 71.6835, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 21.3626651763916, |
| "learning_rate": 0.00015920000000000002, |
| "loss": 56.4135, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 21.728944778442383, |
| "learning_rate": 0.00023920000000000001, |
| "loss": 53.2586, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 16.323501586914062, |
| "learning_rate": 0.0003192, |
| "loss": 51.3419, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 11.02919864654541, |
| "learning_rate": 0.0003992, |
| "loss": 49.8526, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 18.37352180480957, |
| "learning_rate": 0.00047920000000000005, |
| "loss": 48.3665, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 13.519577026367188, |
| "learning_rate": 0.0005592, |
| "loss": 47.4206, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 20.760446548461914, |
| "learning_rate": 0.0006392, |
| "loss": 46.6753, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 25.183908462524414, |
| "learning_rate": 0.0007191999999999999, |
| "loss": 46.4672, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 8.60799789428711, |
| "learning_rate": 0.0007992, |
| "loss": 45.7788, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_accuracy": 0.17874951076320938, |
| "eval_loss": 45.424434661865234, |
| "eval_runtime": 18.125, |
| "eval_samples_per_second": 6.897, |
| "eval_steps_per_second": 0.11, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 7.3993916511535645, |
| "learning_rate": 0.0008792, |
| "loss": 44.8989, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 11.284933090209961, |
| "learning_rate": 0.0009592000000000001, |
| "loss": 44.3576, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 4.662988185882568, |
| "learning_rate": 0.0010391999999999999, |
| "loss": 43.4894, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 7.507369041442871, |
| "learning_rate": 0.0011192, |
| "loss": 42.9519, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 6.868042469024658, |
| "learning_rate": 0.0011992, |
| "loss": 42.3116, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 10.319790840148926, |
| "learning_rate": 0.0012791999999999999, |
| "loss": 43.2361, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 5.94151496887207, |
| "learning_rate": 0.0013592, |
| "loss": 42.3295, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 6.9707865715026855, |
| "learning_rate": 0.0014392, |
| "loss": 42.1999, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 17.769460678100586, |
| "learning_rate": 0.0015192, |
| "loss": 41.7166, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 3.8550658226013184, |
| "learning_rate": 0.0015992, |
| "loss": 41.1934, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.20908414872798434, |
| "eval_loss": 41.37644958496094, |
| "eval_runtime": 3.4901, |
| "eval_samples_per_second": 35.816, |
| "eval_steps_per_second": 0.573, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 3.47354793548584, |
| "learning_rate": 0.0016792, |
| "loss": 40.8328, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 2.9095890522003174, |
| "learning_rate": 0.0017592, |
| "loss": 39.7859, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 4.170510768890381, |
| "learning_rate": 0.0018392, |
| "loss": 40.0358, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 9.68923282623291, |
| "learning_rate": 0.0019192, |
| "loss": 40.5113, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 4.442409992218018, |
| "learning_rate": 0.0019992, |
| "loss": 40.2312, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 3.0799851417541504, |
| "learning_rate": 0.001999978563623903, |
| "loss": 39.559, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 2.932781219482422, |
| "learning_rate": 0.0019999133871331223, |
| "loss": 39.2884, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 3.138781785964966, |
| "learning_rate": 0.0019998044711915177, |
| "loss": 38.8602, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 2.664058208465576, |
| "learning_rate": 0.0019996518205634257, |
| "loss": 37.9913, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 2.482013702392578, |
| "learning_rate": 0.0019994554419262797, |
| "loss": 37.4244, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.2354109589041096, |
| "eval_loss": 38.45709991455078, |
| "eval_runtime": 3.1616, |
| "eval_samples_per_second": 39.537, |
| "eval_steps_per_second": 0.633, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 3.9223365783691406, |
| "learning_rate": 0.001999215343870317, |
| "loss": 38.5841, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 2.736224412918091, |
| "learning_rate": 0.0019989315368982054, |
| "loss": 38.4039, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 2.4935879707336426, |
| "learning_rate": 0.0019986040334245797, |
| "loss": 38.2432, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 3.567474842071533, |
| "learning_rate": 0.001998232847775504, |
| "loss": 38.0513, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 12.996139526367188, |
| "learning_rate": 0.0019978179961878404, |
| "loss": 37.0178, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 3.0554473400115967, |
| "learning_rate": 0.001997359496808541, |
| "loss": 36.8567, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 1.983723759651184, |
| "learning_rate": 0.001996857369693855, |
| "loss": 36.3819, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 1.924381136894226, |
| "learning_rate": 0.0019963116368084486, |
| "loss": 36.6986, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 1.8461555242538452, |
| "learning_rate": 0.001995722322024446, |
| "loss": 37.1027, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 2.570404529571533, |
| "learning_rate": 0.001995089451120385, |
| "loss": 36.6583, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.2556536203522505, |
| "eval_loss": 36.39490509033203, |
| "eval_runtime": 3.1931, |
| "eval_samples_per_second": 39.147, |
| "eval_steps_per_second": 0.626, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.082, |
| "grad_norm": 17.625349044799805, |
| "learning_rate": 0.0019944130517800893, |
| "loss": 36.6922, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.084, |
| "grad_norm": 2.45036244392395, |
| "learning_rate": 0.001993693153591457, |
| "loss": 36.347, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.086, |
| "grad_norm": 3.7300970554351807, |
| "learning_rate": 0.0019929297880451674, |
| "loss": 35.9307, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.088, |
| "grad_norm": 1.9578310251235962, |
| "learning_rate": 0.0019921229885333023, |
| "loss": 35.2814, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.09, |
| "grad_norm": 2.398409843444824, |
| "learning_rate": 0.001991272790347886, |
| "loss": 35.4533, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.092, |
| "grad_norm": 9.658867835998535, |
| "learning_rate": 0.0019903792306793415, |
| "loss": 36.307, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.094, |
| "grad_norm": 1.590939998626709, |
| "learning_rate": 0.001989442348614863, |
| "loss": 36.1738, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.096, |
| "grad_norm": 1.983271837234497, |
| "learning_rate": 0.0019884621851367075, |
| "loss": 35.8103, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.098, |
| "grad_norm": 2.5749049186706543, |
| "learning_rate": 0.001987438783120401, |
| "loss": 35.6467, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 2.067998170852661, |
| "learning_rate": 0.001986372187332862, |
| "loss": 35.323, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.1, |
| "eval_accuracy": 0.26483953033268104, |
| "eval_loss": 35.489601135253906, |
| "eval_runtime": 3.2889, |
| "eval_samples_per_second": 38.007, |
| "eval_steps_per_second": 0.608, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.102, |
| "grad_norm": 1.8774443864822388, |
| "learning_rate": 0.0019852624444304467, |
| "loss": 34.9978, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.104, |
| "grad_norm": 2.0262346267700195, |
| "learning_rate": 0.0019841096029569044, |
| "loss": 34.6675, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.106, |
| "grad_norm": 1.776671051979065, |
| "learning_rate": 0.0019829137133412556, |
| "loss": 35.4506, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.108, |
| "grad_norm": 1.751766324043274, |
| "learning_rate": 0.001981674827895587, |
| "loss": 35.3834, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.11, |
| "grad_norm": 2.024419069290161, |
| "learning_rate": 0.00198039300081276, |
| "loss": 35.2425, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.112, |
| "grad_norm": 1.8287601470947266, |
| "learning_rate": 0.0019790682881640448, |
| "loss": 35.3261, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.114, |
| "grad_norm": 1.6531343460083008, |
| "learning_rate": 0.001977700747896664, |
| "loss": 34.8158, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.116, |
| "grad_norm": 2.180936098098755, |
| "learning_rate": 0.001976290439831259, |
| "loss": 34.4037, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.118, |
| "grad_norm": 1.750649094581604, |
| "learning_rate": 0.0019748374256592736, |
| "loss": 33.958, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 1.9409301280975342, |
| "learning_rate": 0.0019733417689402543, |
| "loss": 34.2086, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.12, |
| "eval_accuracy": 0.2734716242661448, |
| "eval_loss": 34.69782257080078, |
| "eval_runtime": 3.2919, |
| "eval_samples_per_second": 37.972, |
| "eval_steps_per_second": 0.608, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.122, |
| "grad_norm": 1.9359498023986816, |
| "learning_rate": 0.0019718035350990712, |
| "loss": 35.0183, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.124, |
| "grad_norm": 1.7273503541946411, |
| "learning_rate": 0.0019702227914230566, |
| "loss": 34.6009, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.126, |
| "grad_norm": 4.945074081420898, |
| "learning_rate": 0.001968599607059059, |
| "loss": 34.7242, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 1.6802929639816284, |
| "learning_rate": 0.0019669340530104207, |
| "loss": 34.5441, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.13, |
| "grad_norm": 1.8048967123031616, |
| "learning_rate": 0.001965226202133872, |
| "loss": 34.3155, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.132, |
| "grad_norm": 1.656825065612793, |
| "learning_rate": 0.0019634761291363427, |
| "loss": 33.7591, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.134, |
| "grad_norm": 1.642227053642273, |
| "learning_rate": 0.0019616839105716954, |
| "loss": 33.6263, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.136, |
| "grad_norm": 1.531063199043274, |
| "learning_rate": 0.0019598496248373755, |
| "loss": 34.4644, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.138, |
| "grad_norm": 2.118232011795044, |
| "learning_rate": 0.001957973352170984, |
| "loss": 34.597, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 2.264650344848633, |
| "learning_rate": 0.001956055174646765, |
| "loss": 34.2066, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.14, |
| "eval_accuracy": 0.27988845401174167, |
| "eval_loss": 34.06083679199219, |
| "eval_runtime": 3.2263, |
| "eval_samples_per_second": 38.744, |
| "eval_steps_per_second": 0.62, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.142, |
| "grad_norm": 1.5203992128372192, |
| "learning_rate": 0.0019540951761720174, |
| "loss": 34.176, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.144, |
| "grad_norm": 1.5354284048080444, |
| "learning_rate": 0.0019520934424834247, |
| "loss": 33.8863, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.146, |
| "grad_norm": 1.9357357025146484, |
| "learning_rate": 0.0019500500611433025, |
| "loss": 33.6482, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.148, |
| "grad_norm": 1.6156283617019653, |
| "learning_rate": 0.0019479651215357707, |
| "loss": 33.4445, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 1.962512731552124, |
| "learning_rate": 0.0019458387148628417, |
| "loss": 33.8778, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.152, |
| "grad_norm": 1.920979380607605, |
| "learning_rate": 0.001943670934140432, |
| "loss": 34.0846, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.154, |
| "grad_norm": 1.5394798517227173, |
| "learning_rate": 0.0019414618741942936, |
| "loss": 33.9074, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.156, |
| "grad_norm": 1.8427844047546387, |
| "learning_rate": 0.0019392116316558638, |
| "loss": 33.8625, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.158, |
| "grad_norm": 1.5528762340545654, |
| "learning_rate": 0.001936920304958042, |
| "loss": 33.5134, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 2.177480936050415, |
| "learning_rate": 0.0019345879943308804, |
| "loss": 33.2939, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.16, |
| "eval_accuracy": 0.28487866927592953, |
| "eval_loss": 33.50783920288086, |
| "eval_runtime": 3.2372, |
| "eval_samples_per_second": 38.613, |
| "eval_steps_per_second": 0.618, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.162, |
| "grad_norm": 1.4030160903930664, |
| "learning_rate": 0.0019322148017972016, |
| "loss": 32.8396, |
| "step": 8100 |
| }, |
| { |
| "epoch": 0.164, |
| "grad_norm": 2.127406597137451, |
| "learning_rate": 0.001929800831168135, |
| "loss": 32.9587, |
| "step": 8200 |
| }, |
| { |
| "epoch": 0.166, |
| "grad_norm": 1.437472939491272, |
| "learning_rate": 0.001927346188038576, |
| "loss": 33.9238, |
| "step": 8300 |
| }, |
| { |
| "epoch": 0.168, |
| "grad_norm": 1.5702464580535889, |
| "learning_rate": 0.0019248509797825672, |
| "loss": 33.8778, |
| "step": 8400 |
| }, |
| { |
| "epoch": 0.17, |
| "grad_norm": 1.6056277751922607, |
| "learning_rate": 0.0019223153155486009, |
| "loss": 33.6902, |
| "step": 8500 |
| }, |
| { |
| "epoch": 0.172, |
| "grad_norm": 1.6342443227767944, |
| "learning_rate": 0.0019197393062548454, |
| "loss": 33.2568, |
| "step": 8600 |
| }, |
| { |
| "epoch": 0.174, |
| "grad_norm": 1.5210442543029785, |
| "learning_rate": 0.0019171230645842923, |
| "loss": 33.2864, |
| "step": 8700 |
| }, |
| { |
| "epoch": 0.176, |
| "grad_norm": 2.7911336421966553, |
| "learning_rate": 0.0019144667049798272, |
| "loss": 32.8163, |
| "step": 8800 |
| }, |
| { |
| "epoch": 0.178, |
| "grad_norm": 1.7785000801086426, |
| "learning_rate": 0.0019117703436392253, |
| "loss": 32.6067, |
| "step": 8900 |
| }, |
| { |
| "epoch": 0.18, |
| "grad_norm": 1.7542123794555664, |
| "learning_rate": 0.001909034098510066, |
| "loss": 33.2589, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.18, |
| "eval_accuracy": 0.2880078277886497, |
| "eval_loss": 33.28087615966797, |
| "eval_runtime": 3.2266, |
| "eval_samples_per_second": 38.74, |
| "eval_steps_per_second": 0.62, |
| "step": 9000 |
| }, |
| { |
| "epoch": 0.182, |
| "grad_norm": 1.2243378162384033, |
| "learning_rate": 0.001906258089284576, |
| "loss": 33.5476, |
| "step": 9100 |
| }, |
| { |
| "epoch": 0.184, |
| "grad_norm": 1.503909707069397, |
| "learning_rate": 0.0019034424373943915, |
| "loss": 33.7592, |
| "step": 9200 |
| }, |
| { |
| "epoch": 0.186, |
| "grad_norm": 1.6347205638885498, |
| "learning_rate": 0.0019005872660052478, |
| "loss": 33.2045, |
| "step": 9300 |
| }, |
| { |
| "epoch": 0.188, |
| "grad_norm": 1.5243566036224365, |
| "learning_rate": 0.001897692700011591, |
| "loss": 32.841, |
| "step": 9400 |
| }, |
| { |
| "epoch": 0.19, |
| "grad_norm": 1.8256958723068237, |
| "learning_rate": 0.0018947588660311143, |
| "loss": 32.7556, |
| "step": 9500 |
| }, |
| { |
| "epoch": 0.192, |
| "grad_norm": 1.5017831325531006, |
| "learning_rate": 0.0018917858923992211, |
| "loss": 32.2029, |
| "step": 9600 |
| }, |
| { |
| "epoch": 0.194, |
| "grad_norm": 2.1232316493988037, |
| "learning_rate": 0.0018887739091634085, |
| "loss": 32.9246, |
| "step": 9700 |
| }, |
| { |
| "epoch": 0.196, |
| "grad_norm": 2.373900890350342, |
| "learning_rate": 0.0018857230480775807, |
| "loss": 33.4374, |
| "step": 9800 |
| }, |
| { |
| "epoch": 0.198, |
| "grad_norm": 1.6649442911148071, |
| "learning_rate": 0.0018826334425962855, |
| "loss": 33.1861, |
| "step": 9900 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 2.6526882648468018, |
| "learning_rate": 0.0018795052278688753, |
| "loss": 33.3808, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.2, |
| "eval_accuracy": 0.2909138943248532, |
| "eval_loss": 33.03311538696289, |
| "eval_runtime": 3.1811, |
| "eval_samples_per_second": 39.294, |
| "eval_steps_per_second": 0.629, |
| "step": 10000 |
| }, |
| { |
| "epoch": 0.202, |
| "grad_norm": 1.518142819404602, |
| "learning_rate": 0.0018763385407335963, |
| "loss": 32.7453, |
| "step": 10100 |
| }, |
| { |
| "epoch": 0.204, |
| "grad_norm": 1.908083200454712, |
| "learning_rate": 0.001873133519711602, |
| "loss": 32.9035, |
| "step": 10200 |
| }, |
| { |
| "epoch": 0.206, |
| "grad_norm": 2.4479877948760986, |
| "learning_rate": 0.0018698903050008956, |
| "loss": 32.3298, |
| "step": 10300 |
| }, |
| { |
| "epoch": 0.208, |
| "grad_norm": 1.403601050376892, |
| "learning_rate": 0.0018666090384701947, |
| "loss": 32.0576, |
| "step": 10400 |
| }, |
| { |
| "epoch": 0.21, |
| "grad_norm": 1.7655203342437744, |
| "learning_rate": 0.001863289863652727, |
| "loss": 32.8307, |
| "step": 10500 |
| }, |
| { |
| "epoch": 0.212, |
| "grad_norm": 1.317898154258728, |
| "learning_rate": 0.0018599329257399516, |
| "loss": 32.8206, |
| "step": 10600 |
| }, |
| { |
| "epoch": 0.214, |
| "grad_norm": 1.6275999546051025, |
| "learning_rate": 0.0018565383715752083, |
| "loss": 33.1208, |
| "step": 10700 |
| }, |
| { |
| "epoch": 0.216, |
| "grad_norm": 1.350576639175415, |
| "learning_rate": 0.0018531063496472927, |
| "loss": 33.09, |
| "step": 10800 |
| }, |
| { |
| "epoch": 0.218, |
| "grad_norm": 2.071432590484619, |
| "learning_rate": 0.0018496370100839622, |
| "loss": 32.4823, |
| "step": 10900 |
| }, |
| { |
| "epoch": 0.22, |
| "grad_norm": 1.756463646888733, |
| "learning_rate": 0.0018461305046453683, |
| "loss": 32.3173, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.22, |
| "eval_accuracy": 0.29537964774951075, |
| "eval_loss": 32.61259078979492, |
| "eval_runtime": 3.1999, |
| "eval_samples_per_second": 39.064, |
| "eval_steps_per_second": 0.625, |
| "step": 11000 |
| }, |
| { |
| "epoch": 0.222, |
| "grad_norm": 2.6366186141967773, |
| "learning_rate": 0.0018425869867174187, |
| "loss": 32.0905, |
| "step": 11100 |
| }, |
| { |
| "epoch": 0.224, |
| "grad_norm": 1.3121482133865356, |
| "learning_rate": 0.0018390066113050665, |
| "loss": 32.2553, |
| "step": 11200 |
| }, |
| { |
| "epoch": 0.226, |
| "grad_norm": 8.878124237060547, |
| "learning_rate": 0.0018353895350255317, |
| "loss": 33.1959, |
| "step": 11300 |
| }, |
| { |
| "epoch": 0.228, |
| "grad_norm": 2.0059814453125, |
| "learning_rate": 0.0018317359161014477, |
| "loss": 32.804, |
| "step": 11400 |
| }, |
| { |
| "epoch": 0.23, |
| "grad_norm": 1.3689004182815552, |
| "learning_rate": 0.001828045914353943, |
| "loss": 32.6974, |
| "step": 11500 |
| }, |
| { |
| "epoch": 0.232, |
| "grad_norm": 1.6248316764831543, |
| "learning_rate": 0.0018243196911956476, |
| "loss": 32.5622, |
| "step": 11600 |
| }, |
| { |
| "epoch": 0.234, |
| "grad_norm": 1.3218642473220825, |
| "learning_rate": 0.0018205574096236336, |
| "loss": 32.348, |
| "step": 11700 |
| }, |
| { |
| "epoch": 0.236, |
| "grad_norm": 1.4009215831756592, |
| "learning_rate": 0.0018167592342122857, |
| "loss": 31.677, |
| "step": 11800 |
| }, |
| { |
| "epoch": 0.238, |
| "grad_norm": 1.8492565155029297, |
| "learning_rate": 0.0018129253311061002, |
| "loss": 31.6381, |
| "step": 11900 |
| }, |
| { |
| "epoch": 0.24, |
| "grad_norm": 2.3902571201324463, |
| "learning_rate": 0.0018090558680124193, |
| "loss": 32.4851, |
| "step": 12000 |
| }, |
| { |
| "epoch": 0.24, |
| "eval_accuracy": 0.300720156555773, |
| "eval_loss": 32.26734161376953, |
| "eval_runtime": 3.3297, |
| "eval_samples_per_second": 37.54, |
| "eval_steps_per_second": 0.601, |
| "step": 12000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 8, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|