| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 3933, |
| "global_step": 3934, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0, |
| "eval_loss": 8.882436752319336, |
| "eval_ppl": 7204.32452, |
| "eval_runtime": 251.4297, |
| "eval_samples_per_second": 1161.705, |
| "eval_steps_per_second": 24.206, |
| "memory/device_reserved (GiB)": 40.85, |
| "memory/max_active (GiB)": 40.75, |
| "memory/max_allocated (GiB)": 40.75, |
| "step": 0 |
| }, |
| { |
| "epoch": 0.012710921859607868, |
| "grad_norm": 5.21875, |
| "learning_rate": 0.00019999459941110254, |
| "loss": 6.907210083007812, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 999.45495, |
| "step": 50, |
| "tokens/total": 78626816, |
| "tokens/train_per_sec_per_gpu": 1148.16, |
| "tokens/trainable": 78551592 |
| }, |
| { |
| "epoch": 0.025421843719215735, |
| "grad_norm": 2.15625, |
| "learning_rate": 0.00019997187610487207, |
| "loss": 5.735242919921875, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 309.58817, |
| "step": 100, |
| "tokens/total": 157257728, |
| "tokens/train_per_sec_per_gpu": 1149.32, |
| "tokens/trainable": 157108256 |
| }, |
| { |
| "epoch": 0.038132765578823606, |
| "grad_norm": 1.453125, |
| "learning_rate": 0.00019993140447924126, |
| "loss": 5.411905517578125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 224.05813, |
| "step": 150, |
| "tokens/total": 235888640, |
| "tokens/train_per_sec_per_gpu": 1154.11, |
| "tokens/trainable": 235656864 |
| }, |
| { |
| "epoch": 0.05084368743843147, |
| "grad_norm": 1.25, |
| "learning_rate": 0.00019987319171926423, |
| "loss": 5.301961669921875, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 200.73019, |
| "step": 200, |
| "tokens/total": 314519552, |
| "tokens/train_per_sec_per_gpu": 1152.76, |
| "tokens/trainable": 314208672 |
| }, |
| { |
| "epoch": 0.06355460929803934, |
| "grad_norm": 1.5234375, |
| "learning_rate": 0.00019979724815963408, |
| "loss": 5.258994140625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 192.28798, |
| "step": 250, |
| "tokens/total": 393146368, |
| "tokens/train_per_sec_per_gpu": 738.98, |
| "tokens/trainable": 392753440 |
| }, |
| { |
| "epoch": 0.07626553115764721, |
| "grad_norm": 1.078125, |
| "learning_rate": 0.0001997035872828481, |
| "loss": 5.239044189453125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 188.48986, |
| "step": 300, |
| "tokens/total": 471785472, |
| "tokens/train_per_sec_per_gpu": 1147.68, |
| "tokens/trainable": 471316160 |
| }, |
| { |
| "epoch": 0.08897645301725508, |
| "grad_norm": 1.015625, |
| "learning_rate": 0.00019959222571681436, |
| "loss": 5.224154663085938, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 185.70412, |
| "step": 350, |
| "tokens/total": 550420480, |
| "tokens/train_per_sec_per_gpu": 1149.3, |
| "tokens/trainable": 549873472 |
| }, |
| { |
| "epoch": 0.10168737487686294, |
| "grad_norm": 1.359375, |
| "learning_rate": 0.00019946318323189936, |
| "loss": 5.208404541015625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 182.80217, |
| "step": 400, |
| "tokens/total": 629022720, |
| "tokens/train_per_sec_per_gpu": 1151.54, |
| "tokens/trainable": 628395264 |
| }, |
| { |
| "epoch": 0.11439829673647081, |
| "grad_norm": 1.546875, |
| "learning_rate": 0.00019931648273741856, |
| "loss": 5.199439697265625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 181.1707, |
| "step": 450, |
| "tokens/total": 707637248, |
| "tokens/train_per_sec_per_gpu": 1012.12, |
| "tokens/trainable": 706928384 |
| }, |
| { |
| "epoch": 0.12710921859607868, |
| "grad_norm": 1.8359375, |
| "learning_rate": 0.00019915215027756892, |
| "loss": 5.187381591796875, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 178.99925, |
| "step": 500, |
| "tokens/total": 786255872, |
| "tokens/train_per_sec_per_gpu": 1148.12, |
| "tokens/trainable": 785466688 |
| }, |
| { |
| "epoch": 0.13982014045568655, |
| "grad_norm": 1.75, |
| "learning_rate": 0.00019897021502680525, |
| "loss": 5.182908935546875, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 178.20043, |
| "step": 550, |
| "tokens/total": 864882688, |
| "tokens/train_per_sec_per_gpu": 1150.21, |
| "tokens/trainable": 864011904 |
| }, |
| { |
| "epoch": 0.15253106231529442, |
| "grad_norm": 5.6875, |
| "learning_rate": 0.00019877070928466087, |
| "loss": 5.1762158203125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 177.0117, |
| "step": 600, |
| "tokens/total": 943525888, |
| "tokens/train_per_sec_per_gpu": 1149.34, |
| "tokens/trainable": 942579392 |
| }, |
| { |
| "epoch": 0.1652419841749023, |
| "grad_norm": 1.390625, |
| "learning_rate": 0.00019855366847001328, |
| "loss": 5.169808349609375, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 175.88113, |
| "step": 650, |
| "tokens/total": 1022140416, |
| "tokens/train_per_sec_per_gpu": 1151.84, |
| "tokens/trainable": 1021113408 |
| }, |
| { |
| "epoch": 0.17795290603451017, |
| "grad_norm": 1.2734375, |
| "learning_rate": 0.00019831913111479618, |
| "loss": 5.165040283203125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 175.04451, |
| "step": 700, |
| "tokens/total": 1100767232, |
| "tokens/train_per_sec_per_gpu": 1152.16, |
| "tokens/trainable": 1099662208 |
| }, |
| { |
| "epoch": 0.190663827894118, |
| "grad_norm": 1.75, |
| "learning_rate": 0.00019806713885715875, |
| "loss": 5.161105346679688, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 174.35707, |
| "step": 750, |
| "tokens/total": 1179402240, |
| "tokens/train_per_sec_per_gpu": 1143.29, |
| "tokens/trainable": 1178215296 |
| }, |
| { |
| "epoch": 0.20337474975372588, |
| "grad_norm": 1.3046875, |
| "learning_rate": 0.00019779773643407352, |
| "loss": 5.1522760009765625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 172.82439, |
| "step": 800, |
| "tokens/total": 1258024960, |
| "tokens/train_per_sec_per_gpu": 1140.81, |
| "tokens/trainable": 1256756096 |
| }, |
| { |
| "epoch": 0.21608567161333375, |
| "grad_norm": 0.796875, |
| "learning_rate": 0.00019751097167339406, |
| "loss": 5.149384765625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 172.32544, |
| "step": 850, |
| "tokens/total": 1336647680, |
| "tokens/train_per_sec_per_gpu": 1101.91, |
| "tokens/trainable": 1335300352 |
| }, |
| { |
| "epoch": 0.22879659347294162, |
| "grad_norm": 1.2421875, |
| "learning_rate": 0.00019720689548536398, |
| "loss": 5.145052490234375, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 171.58049, |
| "step": 900, |
| "tokens/total": 1415266304, |
| "tokens/train_per_sec_per_gpu": 1149.04, |
| "tokens/trainable": 1413838848 |
| }, |
| { |
| "epoch": 0.2415075153325495, |
| "grad_norm": 0.73828125, |
| "learning_rate": 0.00019688556185357863, |
| "loss": 5.139707641601563, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 170.66587, |
| "step": 950, |
| "tokens/total": 1493872640, |
| "tokens/train_per_sec_per_gpu": 891.45, |
| "tokens/trainable": 1492370688 |
| }, |
| { |
| "epoch": 0.25421843719215736, |
| "grad_norm": 0.8359375, |
| "learning_rate": 0.00019654702782540126, |
| "loss": 5.133409423828125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 169.59435, |
| "step": 1000, |
| "tokens/total": 1572495360, |
| "tokens/train_per_sec_per_gpu": 1148.59, |
| "tokens/trainable": 1570911360 |
| }, |
| { |
| "epoch": 0.26692935905176524, |
| "grad_norm": 0.9609375, |
| "learning_rate": 0.00019619135350183525, |
| "loss": 5.134027099609375, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 169.69914, |
| "step": 1050, |
| "tokens/total": 1651118080, |
| "tokens/train_per_sec_per_gpu": 1148.45, |
| "tokens/trainable": 1649451776 |
| }, |
| { |
| "epoch": 0.2796402809113731, |
| "grad_norm": 0.87109375, |
| "learning_rate": 0.00019581860202685402, |
| "loss": 5.128356323242188, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 168.73954, |
| "step": 1100, |
| "tokens/total": 1729744896, |
| "tokens/train_per_sec_per_gpu": 690.35, |
| "tokens/trainable": 1727996928 |
| }, |
| { |
| "epoch": 0.292351202770981, |
| "grad_norm": 0.63671875, |
| "learning_rate": 0.00019542883957619122, |
| "loss": 5.123819580078125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 167.97574, |
| "step": 1150, |
| "tokens/total": 1808388096, |
| "tokens/train_per_sec_per_gpu": 881.69, |
| "tokens/trainable": 1806564992 |
| }, |
| { |
| "epoch": 0.30506212463058885, |
| "grad_norm": 1.6328125, |
| "learning_rate": 0.00019502213534559198, |
| "loss": 5.119311828613281, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 167.22025, |
| "step": 1200, |
| "tokens/total": 1887014912, |
| "tokens/train_per_sec_per_gpu": 1140.25, |
| "tokens/trainable": 1885116160 |
| }, |
| { |
| "epoch": 0.3177730464901967, |
| "grad_norm": 0.9921875, |
| "learning_rate": 0.00019459856153852863, |
| "loss": 5.118416137695313, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 167.07054, |
| "step": 1250, |
| "tokens/total": 1965645824, |
| "tokens/train_per_sec_per_gpu": 1141.31, |
| "tokens/trainable": 1963666176 |
| }, |
| { |
| "epoch": 0.3304839683498046, |
| "grad_norm": 1.7109375, |
| "learning_rate": 0.000194158193353382, |
| "loss": 5.116240844726563, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 166.70751, |
| "step": 1300, |
| "tokens/total": 2044252160, |
| "tokens/train_per_sec_per_gpu": 1148.85, |
| "tokens/trainable": 2042193664 |
| }, |
| { |
| "epoch": 0.34319489020941246, |
| "grad_norm": 1.1875, |
| "learning_rate": 0.0001937011089700914, |
| "loss": 5.1140374755859375, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 166.3406, |
| "step": 1350, |
| "tokens/total": 2122862592, |
| "tokens/train_per_sec_per_gpu": 1143.78, |
| "tokens/trainable": 2120723840 |
| }, |
| { |
| "epoch": 0.35590581206902033, |
| "grad_norm": 0.95703125, |
| "learning_rate": 0.000193227389536275, |
| "loss": 5.111378479003906, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 165.89889, |
| "step": 1400, |
| "tokens/total": 2201497600, |
| "tokens/train_per_sec_per_gpu": 900.16, |
| "tokens/trainable": 2199277824 |
| }, |
| { |
| "epoch": 0.36861673392862815, |
| "grad_norm": 1.0, |
| "learning_rate": 0.00019273711915282339, |
| "loss": 5.110181274414063, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 165.70039, |
| "step": 1450, |
| "tokens/total": 2280120320, |
| "tokens/train_per_sec_per_gpu": 1153.87, |
| "tokens/trainable": 2277814272 |
| }, |
| { |
| "epoch": 0.381327655788236, |
| "grad_norm": 1.0234375, |
| "learning_rate": 0.00019223038485896896, |
| "loss": 5.105591430664062, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 164.94159, |
| "step": 1500, |
| "tokens/total": 2358730752, |
| "tokens/train_per_sec_per_gpu": 1137.7, |
| "tokens/trainable": 2356336384 |
| }, |
| { |
| "epoch": 0.3940385776478439, |
| "grad_norm": 0.890625, |
| "learning_rate": 0.00019170727661683358, |
| "loss": 5.106539916992188, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 165.09811, |
| "step": 1550, |
| "tokens/total": 2437357568, |
| "tokens/train_per_sec_per_gpu": 1150.48, |
| "tokens/trainable": 2434874624 |
| }, |
| { |
| "epoch": 0.40674949950745176, |
| "grad_norm": 1.6796875, |
| "learning_rate": 0.00019116788729545728, |
| "loss": 5.101802978515625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 164.3179, |
| "step": 1600, |
| "tokens/total": 2515980288, |
| "tokens/train_per_sec_per_gpu": 1008.97, |
| "tokens/trainable": 2513408512 |
| }, |
| { |
| "epoch": 0.41946042136705963, |
| "grad_norm": 0.76953125, |
| "learning_rate": 0.00019061231265431086, |
| "loss": 5.1004248046875, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 164.0916, |
| "step": 1650, |
| "tokens/total": 2594619392, |
| "tokens/train_per_sec_per_gpu": 996.46, |
| "tokens/trainable": 2591954176 |
| }, |
| { |
| "epoch": 0.4321713432266675, |
| "grad_norm": 1.015625, |
| "learning_rate": 0.0001900406513262956, |
| "loss": 5.095927734375, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 163.35532, |
| "step": 1700, |
| "tokens/total": 2673242112, |
| "tokens/train_per_sec_per_gpu": 1143.08, |
| "tokens/trainable": 2670494464 |
| }, |
| { |
| "epoch": 0.4448822650862754, |
| "grad_norm": 1.0625, |
| "learning_rate": 0.0001894530048002325, |
| "loss": 5.0997857666015625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 163.98677, |
| "step": 1750, |
| "tokens/total": 2751873024, |
| "tokens/train_per_sec_per_gpu": 1147.88, |
| "tokens/trainable": 2749035264 |
| }, |
| { |
| "epoch": 0.45759318694588325, |
| "grad_norm": 1.171875, |
| "learning_rate": 0.00018884947740284465, |
| "loss": 5.0969171142578125, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 163.51703, |
| "step": 1800, |
| "tokens/total": 2830499840, |
| "tokens/train_per_sec_per_gpu": 1003.19, |
| "tokens/trainable": 2827570944 |
| }, |
| { |
| "epoch": 0.4703041088054911, |
| "grad_norm": 0.875, |
| "learning_rate": 0.00018823017628023588, |
| "loss": 5.09571044921875, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 163.31983, |
| "step": 1850, |
| "tokens/total": 2909130752, |
| "tokens/train_per_sec_per_gpu": 899.92, |
| "tokens/trainable": 2906111744 |
| }, |
| { |
| "epoch": 0.483015030665099, |
| "grad_norm": 0.7890625, |
| "learning_rate": 0.00018759521137886874, |
| "loss": 5.092830200195312, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 162.85011, |
| "step": 1900, |
| "tokens/total": 2987749376, |
| "tokens/train_per_sec_per_gpu": 1144.23, |
| "tokens/trainable": 2984636672 |
| }, |
| { |
| "epoch": 0.49572595252470686, |
| "grad_norm": 1.1171875, |
| "learning_rate": 0.00018694469542604526, |
| "loss": 5.093046264648438, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 162.8853, |
| "step": 1950, |
| "tokens/total": 3066388480, |
| "tokens/train_per_sec_per_gpu": 1139.03, |
| "tokens/trainable": 3063186944 |
| }, |
| { |
| "epoch": 0.5084368743843147, |
| "grad_norm": 0.7578125, |
| "learning_rate": 0.00018627874390989425, |
| "loss": 5.092771301269531, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 162.84052, |
| "step": 2000, |
| "tokens/total": 3145011200, |
| "tokens/train_per_sec_per_gpu": 1148.33, |
| "tokens/trainable": 3141722368 |
| }, |
| { |
| "epoch": 0.5211477962439226, |
| "grad_norm": 0.734375, |
| "learning_rate": 0.0001855974750588684, |
| "loss": 5.088687438964843, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 162.17686, |
| "step": 2050, |
| "tokens/total": 3223629824, |
| "tokens/train_per_sec_per_gpu": 895.72, |
| "tokens/trainable": 3220251904 |
| }, |
| { |
| "epoch": 0.5338587181035305, |
| "grad_norm": 0.94140625, |
| "learning_rate": 0.00018490100982075446, |
| "loss": 5.087530822753906, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.98939, |
| "step": 2100, |
| "tokens/total": 3302248448, |
| "tokens/train_per_sec_per_gpu": 887.24, |
| "tokens/trainable": 3298778880 |
| }, |
| { |
| "epoch": 0.5465696399631383, |
| "grad_norm": 0.83203125, |
| "learning_rate": 0.00018418947184120147, |
| "loss": 5.090943908691406, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 162.54322, |
| "step": 2150, |
| "tokens/total": 3380875264, |
| "tokens/train_per_sec_per_gpu": 1144.27, |
| "tokens/trainable": 3377318656 |
| }, |
| { |
| "epoch": 0.5592805618227462, |
| "grad_norm": 0.98828125, |
| "learning_rate": 0.0001834629874417692, |
| "loss": 5.08627197265625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.7856, |
| "step": 2200, |
| "tokens/total": 3459510272, |
| "tokens/train_per_sec_per_gpu": 1141.64, |
| "tokens/trainable": 3455864576 |
| }, |
| { |
| "epoch": 0.5719914836823541, |
| "grad_norm": 1.640625, |
| "learning_rate": 0.00018272168559750206, |
| "loss": 5.087280883789062, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.9489, |
| "step": 2250, |
| "tokens/total": 3538141184, |
| "tokens/train_per_sec_per_gpu": 1137.59, |
| "tokens/trainable": 3534408704 |
| }, |
| { |
| "epoch": 0.584702405541962, |
| "grad_norm": 0.74609375, |
| "learning_rate": 0.00018196569791403174, |
| "loss": 5.0849853515625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.57757, |
| "step": 2300, |
| "tokens/total": 3616763904, |
| "tokens/train_per_sec_per_gpu": 889.48, |
| "tokens/trainable": 3612942848 |
| }, |
| { |
| "epoch": 0.5974133274015698, |
| "grad_norm": 0.96484375, |
| "learning_rate": 0.0001811951586042128, |
| "loss": 5.08361572265625, |
| "memory/device_reserved (GiB)": 60.46, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.35642, |
| "step": 2350, |
| "tokens/total": 3695378432, |
| "tokens/train_per_sec_per_gpu": 1140.81, |
| "tokens/trainable": 3691470080 |
| }, |
| { |
| "epoch": 0.6101242492611777, |
| "grad_norm": 0.87890625, |
| "learning_rate": 0.00018041020446429547, |
| "loss": 5.084944152832032, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.57092, |
| "step": 2400, |
| "tokens/total": 3774013440, |
| "tokens/train_per_sec_per_gpu": 1147.78, |
| "tokens/trainable": 3770012672 |
| }, |
| { |
| "epoch": 0.6228351711207856, |
| "grad_norm": 0.984375, |
| "learning_rate": 0.0001796109748496398, |
| "loss": 5.079226684570313, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.64977, |
| "step": 2450, |
| "tokens/total": 3852640256, |
| "tokens/train_per_sec_per_gpu": 1148.28, |
| "tokens/trainable": 3848548096 |
| }, |
| { |
| "epoch": 0.6355460929803934, |
| "grad_norm": 1.1171875, |
| "learning_rate": 0.00017879761164997553, |
| "loss": 5.082112426757813, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.11404, |
| "step": 2500, |
| "tokens/total": 3931279360, |
| "tokens/train_per_sec_per_gpu": 1145.45, |
| "tokens/trainable": 3927099136 |
| }, |
| { |
| "epoch": 0.6482570148400013, |
| "grad_norm": 1.2109375, |
| "learning_rate": 0.00017797025926421172, |
| "loss": 5.081408081054687, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 161.0006, |
| "step": 2550, |
| "tokens/total": 4009910272, |
| "tokens/train_per_sec_per_gpu": 995.23, |
| "tokens/trainable": 4005641472 |
| }, |
| { |
| "epoch": 0.6609679366996092, |
| "grad_norm": 0.80859375, |
| "learning_rate": 0.0001771290645748015, |
| "loss": 5.0802920532226565, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.82102, |
| "step": 2600, |
| "tokens/total": 4088532992, |
| "tokens/train_per_sec_per_gpu": 1141.69, |
| "tokens/trainable": 4084177152 |
| }, |
| { |
| "epoch": 0.673678858559217, |
| "grad_norm": 1.03125, |
| "learning_rate": 0.00017627417692166528, |
| "loss": 5.079031066894531, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.61835, |
| "step": 2650, |
| "tokens/total": 4167155712, |
| "tokens/train_per_sec_per_gpu": 1138.96, |
| "tokens/trainable": 4162711296 |
| }, |
| { |
| "epoch": 0.6863897804188249, |
| "grad_norm": 1.265625, |
| "learning_rate": 0.00017540574807567818, |
| "loss": 5.079642639160157, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.71661, |
| "step": 2700, |
| "tokens/total": 4245798912, |
| "tokens/train_per_sec_per_gpu": 1144.53, |
| "tokens/trainable": 4241262592 |
| }, |
| { |
| "epoch": 0.6991007022784328, |
| "grad_norm": 0.69140625, |
| "learning_rate": 0.0001745239322117255, |
| "loss": 5.078349304199219, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.50889, |
| "step": 2750, |
| "tokens/total": 4324409344, |
| "tokens/train_per_sec_per_gpu": 892.26, |
| "tokens/trainable": 4319796224 |
| }, |
| { |
| "epoch": 0.7118116241380407, |
| "grad_norm": 0.90234375, |
| "learning_rate": 0.0001736288858813317, |
| "loss": 5.078107299804688, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.47005, |
| "step": 2800, |
| "tokens/total": 4403052544, |
| "tokens/train_per_sec_per_gpu": 984.67, |
| "tokens/trainable": 4398406144 |
| }, |
| { |
| "epoch": 0.7245225459976484, |
| "grad_norm": 0.8359375, |
| "learning_rate": 0.00017272076798486726, |
| "loss": 5.076226806640625, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.16857, |
| "step": 2850, |
| "tokens/total": 4481658880, |
| "tokens/train_per_sec_per_gpu": 1142.23, |
| "tokens/trainable": 4476985344 |
| }, |
| { |
| "epoch": 0.7372334678572563, |
| "grad_norm": 0.70703125, |
| "learning_rate": 0.00017179973974333857, |
| "loss": 5.076329650878907, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.18504, |
| "step": 2900, |
| "tokens/total": 4560277504, |
| "tokens/train_per_sec_per_gpu": 1143.25, |
| "tokens/trainable": 4555567616 |
| }, |
| { |
| "epoch": 0.7499443897168642, |
| "grad_norm": 0.890625, |
| "learning_rate": 0.00017086596466976594, |
| "loss": 5.074002990722656, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.81278, |
| "step": 2950, |
| "tokens/total": 4638900224, |
| "tokens/train_per_sec_per_gpu": 1007.19, |
| "tokens/trainable": 4634149888 |
| }, |
| { |
| "epoch": 0.762655311576472, |
| "grad_norm": 0.859375, |
| "learning_rate": 0.00016991960854015464, |
| "loss": 5.075715942382812, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 160.08676, |
| "step": 3000, |
| "tokens/total": 4717506560, |
| "tokens/train_per_sec_per_gpu": 1142.63, |
| "tokens/trainable": 4712714752 |
| }, |
| { |
| "epoch": 0.7753662334360799, |
| "grad_norm": 0.984375, |
| "learning_rate": 0.00016896083936406394, |
| "loss": 5.0732049560546875, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.68529, |
| "step": 3050, |
| "tokens/total": 4796145664, |
| "tokens/train_per_sec_per_gpu": 1140.38, |
| "tokens/trainable": 4791320576 |
| }, |
| { |
| "epoch": 0.7880771552956878, |
| "grad_norm": 0.83203125, |
| "learning_rate": 0.00016798982735478023, |
| "loss": 5.072262573242187, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.53488, |
| "step": 3100, |
| "tokens/total": 4874768384, |
| "tokens/train_per_sec_per_gpu": 1010.94, |
| "tokens/trainable": 4869914112 |
| }, |
| { |
| "epoch": 0.8007880771552957, |
| "grad_norm": 1.1015625, |
| "learning_rate": 0.00016700674489909817, |
| "loss": 5.071590881347657, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.42776, |
| "step": 3150, |
| "tokens/total": 4953391104, |
| "tokens/train_per_sec_per_gpu": 1134.74, |
| "tokens/trainable": 4948505600 |
| }, |
| { |
| "epoch": 0.8134989990149035, |
| "grad_norm": 1.3828125, |
| "learning_rate": 0.00016601176652671654, |
| "loss": 5.071939392089844, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.48333, |
| "step": 3200, |
| "tokens/total": 5032034304, |
| "tokens/train_per_sec_per_gpu": 998.75, |
| "tokens/trainable": 5027120640 |
| }, |
| { |
| "epoch": 0.8262099208745114, |
| "grad_norm": 0.77734375, |
| "learning_rate": 0.00016500506887925332, |
| "loss": 5.070206298828125, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.20717, |
| "step": 3250, |
| "tokens/total": 5110657024, |
| "tokens/train_per_sec_per_gpu": 987.57, |
| "tokens/trainable": 5105710592 |
| }, |
| { |
| "epoch": 0.8389208427341193, |
| "grad_norm": 0.6484375, |
| "learning_rate": 0.000163986830678886, |
| "loss": 5.072908325195312, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.63793, |
| "step": 3300, |
| "tokens/total": 5189275648, |
| "tokens/train_per_sec_per_gpu": 1137.51, |
| "tokens/trainable": 5184292864 |
| }, |
| { |
| "epoch": 0.8516317645937271, |
| "grad_norm": 0.94140625, |
| "learning_rate": 0.00016295723269662247, |
| "loss": 5.069717712402344, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.1294, |
| "step": 3350, |
| "tokens/total": 5267918848, |
| "tokens/train_per_sec_per_gpu": 1141.74, |
| "tokens/trainable": 5262910976 |
| }, |
| { |
| "epoch": 0.864342686453335, |
| "grad_norm": 0.98828125, |
| "learning_rate": 0.00016191645772020827, |
| "loss": 5.071309204101563, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.38286, |
| "step": 3400, |
| "tokens/total": 5346545664, |
| "tokens/train_per_sec_per_gpu": 998.73, |
| "tokens/trainable": 5341506560 |
| }, |
| { |
| "epoch": 0.8770536083129429, |
| "grad_norm": 1.21875, |
| "learning_rate": 0.00016086469052167547, |
| "loss": 5.069463500976562, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.08895, |
| "step": 3450, |
| "tokens/total": 5425172480, |
| "tokens/train_per_sec_per_gpu": 990.05, |
| "tokens/trainable": 5420105216 |
| }, |
| { |
| "epoch": 0.8897645301725507, |
| "grad_norm": 0.67578125, |
| "learning_rate": 0.00015980211782453977, |
| "loss": 5.070612182617188, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.2718, |
| "step": 3500, |
| "tokens/total": 5503782912, |
| "tokens/train_per_sec_per_gpu": 1139.16, |
| "tokens/trainable": 5498675200 |
| }, |
| { |
| "epoch": 0.9024754520321586, |
| "grad_norm": 0.8984375, |
| "learning_rate": 0.0001587289282706508, |
| "loss": 5.069385375976562, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.07653, |
| "step": 3550, |
| "tokens/total": 5582409728, |
| "tokens/train_per_sec_per_gpu": 991.04, |
| "tokens/trainable": 5577267712 |
| }, |
| { |
| "epoch": 0.9151863738917665, |
| "grad_norm": 1.0078125, |
| "learning_rate": 0.00015764531238670175, |
| "loss": 5.068446044921875, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 158.92717, |
| "step": 3600, |
| "tokens/total": 5661048832, |
| "tokens/train_per_sec_per_gpu": 1146.26, |
| "tokens/trainable": 5655876608 |
| }, |
| { |
| "epoch": 0.9278972957513744, |
| "grad_norm": 1.09375, |
| "learning_rate": 0.00015655146255040501, |
| "loss": 5.070053405761719, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.18283, |
| "step": 3650, |
| "tokens/total": 5739687936, |
| "tokens/train_per_sec_per_gpu": 1143.44, |
| "tokens/trainable": 5734480384 |
| }, |
| { |
| "epoch": 0.9406082176109822, |
| "grad_norm": 0.71484375, |
| "learning_rate": 0.00015544757295633837, |
| "loss": 5.067489013671875, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 158.77514, |
| "step": 3700, |
| "tokens/total": 5818314752, |
| "tokens/train_per_sec_per_gpu": 1016.18, |
| "tokens/trainable": 5813072896 |
| }, |
| { |
| "epoch": 0.9533191394705901, |
| "grad_norm": 0.73046875, |
| "learning_rate": 0.00015433383958146906, |
| "loss": 5.07120849609375, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 159.36681, |
| "step": 3750, |
| "tokens/total": 5896941568, |
| "tokens/train_per_sec_per_gpu": 1158.51, |
| "tokens/trainable": 5891662848 |
| }, |
| { |
| "epoch": 0.966030061330198, |
| "grad_norm": 0.8359375, |
| "learning_rate": 0.00015321046015036167, |
| "loss": 5.0687188720703125, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 158.97054, |
| "step": 3800, |
| "tokens/total": 5975556096, |
| "tokens/train_per_sec_per_gpu": 1155.48, |
| "tokens/trainable": 5970243072 |
| }, |
| { |
| "epoch": 0.9787409831898058, |
| "grad_norm": 1.2421875, |
| "learning_rate": 0.0001520776341000751, |
| "loss": 5.067420959472656, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 158.76434, |
| "step": 3850, |
| "tokens/total": 6054170624, |
| "tokens/train_per_sec_per_gpu": 1157.96, |
| "tokens/trainable": 6048813568 |
| }, |
| { |
| "epoch": 0.9914519050494137, |
| "grad_norm": 0.78125, |
| "learning_rate": 0.00015093556254475618, |
| "loss": 5.067011108398438, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "ppl": 158.69928, |
| "step": 3900, |
| "tokens/total": 6132797440, |
| "tokens/train_per_sec_per_gpu": 1153.86, |
| "tokens/trainable": 6127400960 |
| }, |
| { |
| "epoch": 0.9998411134767549, |
| "eval_loss": 5.067268371582031, |
| "eval_ppl": 158.74012, |
| "eval_runtime": 255.6347, |
| "eval_samples_per_second": 1142.595, |
| "eval_steps_per_second": 23.807, |
| "memory/device_reserved (GiB)": 60.62, |
| "memory/max_active (GiB)": 52.53, |
| "memory/max_allocated (GiB)": 52.53, |
| "step": 3933 |
| } |
| ], |
| "logging_steps": 50, |
| "max_steps": 11799, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1.3283000173283246e+19, |
| "train_batch_size": 6, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|