| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.16, |
| "eval_steps": 1000, |
| "global_step": 8000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": false, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 4.729869365692139, |
| "learning_rate": 0.00033, |
| "loss": 64.03521484375, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 7.427474021911621, |
| "learning_rate": 0.0004999988080137436, |
| "loss": 50.4660400390625, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 499.43450927734375, |
| "learning_rate": 0.0004999889782951015, |
| "loss": 55.620224609375, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 5.543099880218506, |
| "learning_rate": 0.0004999692199574857, |
| "loss": 55.512568359375, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 3.913767099380493, |
| "learning_rate": 0.0004999395337856224, |
| "loss": 54.5164111328125, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 3.7103848457336426, |
| "learning_rate": 0.0004998999209585345, |
| "loss": 48.3706103515625, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 27.09828758239746, |
| "learning_rate": 0.0004998503830494941, |
| "loss": 48.7948681640625, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 357.75579833984375, |
| "learning_rate": 0.0004997909220259598, |
| "loss": 47.264482421875, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 21.32952117919922, |
| "learning_rate": 0.0004997215402494993, |
| "loss": 46.945810546875, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 1339.14501953125, |
| "learning_rate": 0.0004996422404756948, |
| "loss": 50.6924609375, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_loss": 62.66343307495117, |
| "eval_runtime": 11.6524, |
| "eval_samples_per_second": 1.716, |
| "eval_steps_per_second": 0.086, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 29.14947509765625, |
| "learning_rate": 0.0004995530258540343, |
| "loss": 49.645908203125, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 199.71112060546875, |
| "learning_rate": 0.0004994538999277859, |
| "loss": 46.0685693359375, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 26.30093002319336, |
| "learning_rate": 0.0004993448666338572, |
| "loss": 45.418486328125, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 72.15017700195312, |
| "learning_rate": 0.0004992259303026396, |
| "loss": 46.6232666015625, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 16.536649703979492, |
| "learning_rate": 0.0004990970956578351, |
| "loss": 52.807451171875, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 75235.25, |
| "learning_rate": 0.0004989583678162697, |
| "loss": 58.359990234375, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 7981.73974609375, |
| "learning_rate": 0.0004988097522876898, |
| "loss": 58.0731103515625, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 10026.5595703125, |
| "learning_rate": 0.0004986512549745438, |
| "loss": 59.05083984375, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 534.1368408203125, |
| "learning_rate": 0.0004984828821717466, |
| "loss": 54.8443310546875, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 423.7434997558594, |
| "learning_rate": 0.0004983046405664306, |
| "loss": 54.6559326171875, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_loss": 55.864173889160156, |
| "eval_runtime": 1.8383, |
| "eval_samples_per_second": 10.88, |
| "eval_steps_per_second": 0.544, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 7411.40234375, |
| "learning_rate": 0.0004981165372376802, |
| "loss": 58.2853564453125, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 33.24531173706055, |
| "learning_rate": 0.0004979185796562494, |
| "loss": 57.3351123046875, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 1044.70458984375, |
| "learning_rate": 0.0004977107756842668, |
| "loss": 55.583466796875, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 488.1427001953125, |
| "learning_rate": 0.0004974931335749219, |
| "loss": 53.566806640625, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 949.36181640625, |
| "learning_rate": 0.000497265661972138, |
| "loss": 52.38724609375, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 1916.361328125, |
| "learning_rate": 0.0004970283699102291, |
| "loss": 55.593505859375, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 40184.4609375, |
| "learning_rate": 0.0004967812668135405, |
| "loss": 59.1807861328125, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 1653.9197998046875, |
| "learning_rate": 0.0004965243624960747, |
| "loss": 60.0856396484375, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 570.0091552734375, |
| "learning_rate": 0.0004962576671611021, |
| "loss": 55.3298486328125, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 767297.5625, |
| "learning_rate": 0.000495981191400755, |
| "loss": 59.248984375, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_loss": 65.7476806640625, |
| "eval_runtime": 1.8469, |
| "eval_samples_per_second": 10.829, |
| "eval_steps_per_second": 0.541, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 1796894.75, |
| "learning_rate": 0.0004956949461956074, |
| "loss": 61.4278955078125, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 372332.40625, |
| "learning_rate": 0.0004953989429142387, |
| "loss": 58.2212158203125, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 839582.25, |
| "learning_rate": 0.0004950931933127826, |
| "loss": 60.9759326171875, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 456507.9375, |
| "learning_rate": 0.0004947777095344596, |
| "loss": 60.5664697265625, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 42298.0859375, |
| "learning_rate": 0.0004944525041090948, |
| "loss": 59.9312353515625, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 44746.13671875, |
| "learning_rate": 0.0004941175899526208, |
| "loss": 60.0662353515625, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 6649.8818359375, |
| "learning_rate": 0.0004937729803665643, |
| "loss": 61.2509716796875, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 15284.4833984375, |
| "learning_rate": 0.0004934186890375175, |
| "loss": 59.5713232421875, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 13539.5712890625, |
| "learning_rate": 0.0004930547300365956, |
| "loss": 58.5672265625, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 53576.8984375, |
| "learning_rate": 0.0004926811178188765, |
| "loss": 58.3544189453125, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_loss": 58.579750061035156, |
| "eval_runtime": 1.8442, |
| "eval_samples_per_second": 10.845, |
| "eval_steps_per_second": 0.542, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.082, |
| "grad_norm": 160393.71875, |
| "learning_rate": 0.000492297867222828, |
| "loss": 59.3321435546875, |
| "step": 4100 |
| }, |
| { |
| "epoch": 0.084, |
| "grad_norm": 12930.431640625, |
| "learning_rate": 0.0004919049934697177, |
| "loss": 60.8889404296875, |
| "step": 4200 |
| }, |
| { |
| "epoch": 0.086, |
| "grad_norm": 4836.6787109375, |
| "learning_rate": 0.0004915025121630086, |
| "loss": 61.493935546875, |
| "step": 4300 |
| }, |
| { |
| "epoch": 0.088, |
| "grad_norm": 595.083984375, |
| "learning_rate": 0.0004910904392877396, |
| "loss": 58.8386474609375, |
| "step": 4400 |
| }, |
| { |
| "epoch": 0.09, |
| "grad_norm": 12180.2373046875, |
| "learning_rate": 0.0004906687912098906, |
| "loss": 63.4604150390625, |
| "step": 4500 |
| }, |
| { |
| "epoch": 0.092, |
| "grad_norm": 414.4292297363281, |
| "learning_rate": 0.0004902375846757322, |
| "loss": 59.018310546875, |
| "step": 4600 |
| }, |
| { |
| "epoch": 0.094, |
| "grad_norm": 2297.899169921875, |
| "learning_rate": 0.0004897968368111611, |
| "loss": 57.6424853515625, |
| "step": 4700 |
| }, |
| { |
| "epoch": 0.096, |
| "grad_norm": 2821.689697265625, |
| "learning_rate": 0.0004893465651210193, |
| "loss": 58.3983154296875, |
| "step": 4800 |
| }, |
| { |
| "epoch": 0.098, |
| "grad_norm": 2559.970947265625, |
| "learning_rate": 0.0004888867874883995, |
| "loss": 57.3033935546875, |
| "step": 4900 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 8444.46484375, |
| "learning_rate": 0.0004884175221739343, |
| "loss": 56.6698193359375, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.1, |
| "eval_loss": 57.39847946166992, |
| "eval_runtime": 1.8416, |
| "eval_samples_per_second": 10.86, |
| "eval_steps_per_second": 0.543, |
| "step": 5000 |
| }, |
| { |
| "epoch": 0.102, |
| "grad_norm": 6454.49658203125, |
| "learning_rate": 0.0004879387878150716, |
| "loss": 56.9357958984375, |
| "step": 5100 |
| }, |
| { |
| "epoch": 0.104, |
| "grad_norm": 337538.0, |
| "learning_rate": 0.00048745060342533363, |
| "loss": 56.982119140625, |
| "step": 5200 |
| }, |
| { |
| "epoch": 0.106, |
| "grad_norm": 68615.4609375, |
| "learning_rate": 0.0004869529883935625, |
| "loss": 56.92587890625, |
| "step": 5300 |
| }, |
| { |
| "epoch": 0.108, |
| "grad_norm": 116072.59375, |
| "learning_rate": 0.00048644596248314967, |
| "loss": 57.190751953125, |
| "step": 5400 |
| }, |
| { |
| "epoch": 0.11, |
| "grad_norm": 291359.625, |
| "learning_rate": 0.0004859295458312511, |
| "loss": 58.6766162109375, |
| "step": 5500 |
| }, |
| { |
| "epoch": 0.112, |
| "grad_norm": 89412.578125, |
| "learning_rate": 0.0004854037589479878, |
| "loss": 58.688193359375, |
| "step": 5600 |
| }, |
| { |
| "epoch": 0.114, |
| "grad_norm": 293593.4375, |
| "learning_rate": 0.0004848686227156309, |
| "loss": 59.063056640625, |
| "step": 5700 |
| }, |
| { |
| "epoch": 0.116, |
| "grad_norm": 30479.330078125, |
| "learning_rate": 0.0004843241583877724, |
| "loss": 58.46580078125, |
| "step": 5800 |
| }, |
| { |
| "epoch": 0.118, |
| "grad_norm": 89906.7734375, |
| "learning_rate": 0.000483770387588481, |
| "loss": 58.2409228515625, |
| "step": 5900 |
| }, |
| { |
| "epoch": 0.12, |
| "grad_norm": 630418.5, |
| "learning_rate": 0.00048320733231144354, |
| "loss": 58.333916015625, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.12, |
| "eval_loss": 59.336669921875, |
| "eval_runtime": 1.8507, |
| "eval_samples_per_second": 10.807, |
| "eval_steps_per_second": 0.54, |
| "step": 6000 |
| }, |
| { |
| "epoch": 0.122, |
| "grad_norm": 462631.5, |
| "learning_rate": 0.000482635014919091, |
| "loss": 59.005234375, |
| "step": 6100 |
| }, |
| { |
| "epoch": 0.124, |
| "grad_norm": 273326.59375, |
| "learning_rate": 0.00048205345814171074, |
| "loss": 58.9213232421875, |
| "step": 6200 |
| }, |
| { |
| "epoch": 0.126, |
| "grad_norm": 36951.60546875, |
| "learning_rate": 0.00048146268507654377, |
| "loss": 58.82025390625, |
| "step": 6300 |
| }, |
| { |
| "epoch": 0.128, |
| "grad_norm": 103255.4765625, |
| "learning_rate": 0.00048086271918686716, |
| "loss": 58.84619140625, |
| "step": 6400 |
| }, |
| { |
| "epoch": 0.13, |
| "grad_norm": 102904.96875, |
| "learning_rate": 0.00048025358430106227, |
| "loss": 58.514482421875, |
| "step": 6500 |
| }, |
| { |
| "epoch": 0.132, |
| "grad_norm": 368401.625, |
| "learning_rate": 0.00047963530461166826, |
| "loss": 58.3533984375, |
| "step": 6600 |
| }, |
| { |
| "epoch": 0.134, |
| "grad_norm": 307524.84375, |
| "learning_rate": 0.0004790079046744218, |
| "loss": 58.44912109375, |
| "step": 6700 |
| }, |
| { |
| "epoch": 0.136, |
| "grad_norm": 218012.390625, |
| "learning_rate": 0.000478371409407281, |
| "loss": 58.45431640625, |
| "step": 6800 |
| }, |
| { |
| "epoch": 0.138, |
| "grad_norm": 1243047.5, |
| "learning_rate": 0.0004777258440894362, |
| "loss": 58.24828125, |
| "step": 6900 |
| }, |
| { |
| "epoch": 0.14, |
| "grad_norm": 158677.6875, |
| "learning_rate": 0.0004770712343603062, |
| "loss": 58.133662109375, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.14, |
| "eval_loss": 58.2567138671875, |
| "eval_runtime": 1.8382, |
| "eval_samples_per_second": 10.88, |
| "eval_steps_per_second": 0.544, |
| "step": 7000 |
| }, |
| { |
| "epoch": 0.142, |
| "grad_norm": 237161.1875, |
| "learning_rate": 0.0004764076062185194, |
| "loss": 58.3987744140625, |
| "step": 7100 |
| }, |
| { |
| "epoch": 0.144, |
| "grad_norm": 60140.4921875, |
| "learning_rate": 0.00047573498602088154, |
| "loss": 58.801220703125, |
| "step": 7200 |
| }, |
| { |
| "epoch": 0.146, |
| "grad_norm": 494529.25, |
| "learning_rate": 0.00047505340048132916, |
| "loss": 58.503544921875, |
| "step": 7300 |
| }, |
| { |
| "epoch": 0.148, |
| "grad_norm": 561465.625, |
| "learning_rate": 0.00047436287666986803, |
| "loss": 58.614169921875, |
| "step": 7400 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 572518.4375, |
| "learning_rate": 0.00047366344201149856, |
| "loss": 58.28416015625, |
| "step": 7500 |
| }, |
| { |
| "epoch": 0.152, |
| "grad_norm": 1078630.125, |
| "learning_rate": 0.0004729551242851264, |
| "loss": 58.4676513671875, |
| "step": 7600 |
| }, |
| { |
| "epoch": 0.154, |
| "grad_norm": 70299.53125, |
| "learning_rate": 0.00047223795162245886, |
| "loss": 58.200947265625, |
| "step": 7700 |
| }, |
| { |
| "epoch": 0.156, |
| "grad_norm": 68182.265625, |
| "learning_rate": 0.0004715119525068883, |
| "loss": 58.55353515625, |
| "step": 7800 |
| }, |
| { |
| "epoch": 0.158, |
| "grad_norm": 82698.640625, |
| "learning_rate": 0.00047077715577236015, |
| "loss": 58.704990234375, |
| "step": 7900 |
| }, |
| { |
| "epoch": 0.16, |
| "grad_norm": 375052.03125, |
| "learning_rate": 0.0004700335906022283, |
| "loss": 58.521640625, |
| "step": 8000 |
| }, |
| { |
| "epoch": 0.16, |
| "eval_loss": 57.7498779296875, |
| "eval_runtime": 1.8363, |
| "eval_samples_per_second": 10.892, |
| "eval_steps_per_second": 0.545, |
| "step": 8000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 5.699606151168e+16, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|