| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 10.0, |
| "eval_steps": 500, |
| "global_step": 250, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.2, |
| "grad_norm": 11.125, |
| "learning_rate": 6.4000000000000006e-06, |
| "loss": -0.011411023139953614, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 12.0625, |
| "learning_rate": 1.4400000000000001e-05, |
| "loss": -0.02921934425830841, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 28.875, |
| "learning_rate": 2.2400000000000002e-05, |
| "loss": -0.02916721999645233, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.8, |
| "grad_norm": 90.0, |
| "learning_rate": 3.0400000000000004e-05, |
| "loss": -0.11180627346038818, |
| "step": 20 |
| }, |
| { |
| "epoch": 1.0, |
| "grad_norm": 236.0, |
| "learning_rate": 3.8400000000000005e-05, |
| "loss": -0.62957763671875, |
| "step": 25 |
| }, |
| { |
| "epoch": 1.2, |
| "grad_norm": 332.0, |
| "learning_rate": 3.9288888888888894e-05, |
| "loss": -2.0658231735229493, |
| "step": 30 |
| }, |
| { |
| "epoch": 1.4, |
| "grad_norm": 398.0, |
| "learning_rate": 3.8400000000000005e-05, |
| "loss": -3.0921772003173826, |
| "step": 35 |
| }, |
| { |
| "epoch": 1.6, |
| "grad_norm": 1584.0, |
| "learning_rate": 3.7511111111111116e-05, |
| "loss": -14.95286865234375, |
| "step": 40 |
| }, |
| { |
| "epoch": 1.8, |
| "grad_norm": 2848.0, |
| "learning_rate": 3.662222222222223e-05, |
| "loss": -43.402197265625, |
| "step": 45 |
| }, |
| { |
| "epoch": 2.0, |
| "grad_norm": 294.0, |
| "learning_rate": 3.573333333333333e-05, |
| "loss": -54.48800048828125, |
| "step": 50 |
| }, |
| { |
| "epoch": 2.2, |
| "grad_norm": 382.0, |
| "learning_rate": 3.4844444444444444e-05, |
| "loss": -62.52164916992187, |
| "step": 55 |
| }, |
| { |
| "epoch": 2.4, |
| "grad_norm": 181.0, |
| "learning_rate": 3.395555555555556e-05, |
| "loss": -69.7565185546875, |
| "step": 60 |
| }, |
| { |
| "epoch": 2.6, |
| "grad_norm": 1384.0, |
| "learning_rate": 3.3066666666666666e-05, |
| "loss": -68.15775756835937, |
| "step": 65 |
| }, |
| { |
| "epoch": 2.8, |
| "grad_norm": 122.5, |
| "learning_rate": 3.217777777777778e-05, |
| "loss": -74.51124877929688, |
| "step": 70 |
| }, |
| { |
| "epoch": 3.0, |
| "grad_norm": 460.0, |
| "learning_rate": 3.1288888888888896e-05, |
| "loss": -74.828271484375, |
| "step": 75 |
| }, |
| { |
| "epoch": 3.2, |
| "grad_norm": 122.0, |
| "learning_rate": 3.0400000000000004e-05, |
| "loss": -72.76103515625, |
| "step": 80 |
| }, |
| { |
| "epoch": 3.4, |
| "grad_norm": 151.0, |
| "learning_rate": 2.951111111111111e-05, |
| "loss": -73.82017822265625, |
| "step": 85 |
| }, |
| { |
| "epoch": 3.6, |
| "grad_norm": 171.0, |
| "learning_rate": 2.8622222222222223e-05, |
| "loss": -77.53640747070312, |
| "step": 90 |
| }, |
| { |
| "epoch": 3.8, |
| "grad_norm": 120.5, |
| "learning_rate": 2.7733333333333338e-05, |
| "loss": -78.69415283203125, |
| "step": 95 |
| }, |
| { |
| "epoch": 4.0, |
| "grad_norm": 112.5, |
| "learning_rate": 2.6844444444444446e-05, |
| "loss": -79.16287841796876, |
| "step": 100 |
| }, |
| { |
| "epoch": 4.2, |
| "grad_norm": 103.5, |
| "learning_rate": 2.5955555555555557e-05, |
| "loss": -78.008203125, |
| "step": 105 |
| }, |
| { |
| "epoch": 4.4, |
| "grad_norm": 186.0, |
| "learning_rate": 2.5066666666666672e-05, |
| "loss": -81.25070190429688, |
| "step": 110 |
| }, |
| { |
| "epoch": 4.6, |
| "grad_norm": 314.0, |
| "learning_rate": 2.417777777777778e-05, |
| "loss": -80.52830200195312, |
| "step": 115 |
| }, |
| { |
| "epoch": 4.8, |
| "grad_norm": 988.0, |
| "learning_rate": 2.328888888888889e-05, |
| "loss": -77.89945068359376, |
| "step": 120 |
| }, |
| { |
| "epoch": 5.0, |
| "grad_norm": 110.5, |
| "learning_rate": 2.2400000000000002e-05, |
| "loss": -79.39564208984375, |
| "step": 125 |
| }, |
| { |
| "epoch": 5.2, |
| "grad_norm": 144.0, |
| "learning_rate": 2.1511111111111114e-05, |
| "loss": -80.58895874023438, |
| "step": 130 |
| }, |
| { |
| "epoch": 5.4, |
| "grad_norm": 113.5, |
| "learning_rate": 2.0622222222222225e-05, |
| "loss": -82.10526123046876, |
| "step": 135 |
| }, |
| { |
| "epoch": 5.6, |
| "grad_norm": 109.0, |
| "learning_rate": 1.9733333333333336e-05, |
| "loss": -80.39559936523438, |
| "step": 140 |
| }, |
| { |
| "epoch": 5.8, |
| "grad_norm": 122.0, |
| "learning_rate": 1.8844444444444444e-05, |
| "loss": -82.06431274414062, |
| "step": 145 |
| }, |
| { |
| "epoch": 6.0, |
| "grad_norm": 111.0, |
| "learning_rate": 1.7955555555555556e-05, |
| "loss": -84.17305297851563, |
| "step": 150 |
| }, |
| { |
| "epoch": 6.2, |
| "grad_norm": 108.5, |
| "learning_rate": 1.706666666666667e-05, |
| "loss": -84.07409057617187, |
| "step": 155 |
| }, |
| { |
| "epoch": 6.4, |
| "grad_norm": 112.5, |
| "learning_rate": 1.617777777777778e-05, |
| "loss": -85.04705200195312, |
| "step": 160 |
| }, |
| { |
| "epoch": 6.6, |
| "grad_norm": 118.0, |
| "learning_rate": 1.528888888888889e-05, |
| "loss": -82.29384155273438, |
| "step": 165 |
| }, |
| { |
| "epoch": 6.8, |
| "grad_norm": 116.5, |
| "learning_rate": 1.4400000000000001e-05, |
| "loss": -85.50375366210938, |
| "step": 170 |
| }, |
| { |
| "epoch": 7.0, |
| "grad_norm": 312.0, |
| "learning_rate": 1.3511111111111112e-05, |
| "loss": -85.09585571289062, |
| "step": 175 |
| }, |
| { |
| "epoch": 7.2, |
| "grad_norm": 158.0, |
| "learning_rate": 1.2622222222222222e-05, |
| "loss": -88.86329956054688, |
| "step": 180 |
| }, |
| { |
| "epoch": 7.4, |
| "grad_norm": 195.0, |
| "learning_rate": 1.1733333333333335e-05, |
| "loss": -88.39931640625, |
| "step": 185 |
| }, |
| { |
| "epoch": 7.6, |
| "grad_norm": 189.0, |
| "learning_rate": 1.0844444444444446e-05, |
| "loss": -90.6859130859375, |
| "step": 190 |
| }, |
| { |
| "epoch": 7.8, |
| "grad_norm": 229.0, |
| "learning_rate": 9.955555555555556e-06, |
| "loss": -93.2937255859375, |
| "step": 195 |
| }, |
| { |
| "epoch": 8.0, |
| "grad_norm": 616.0, |
| "learning_rate": 9.066666666666667e-06, |
| "loss": -93.46704711914063, |
| "step": 200 |
| }, |
| { |
| "epoch": 8.2, |
| "grad_norm": 1056.0, |
| "learning_rate": 8.177777777777779e-06, |
| "loss": -94.84644775390625, |
| "step": 205 |
| }, |
| { |
| "epoch": 8.4, |
| "grad_norm": 176.0, |
| "learning_rate": 7.28888888888889e-06, |
| "loss": -96.15100708007813, |
| "step": 210 |
| }, |
| { |
| "epoch": 8.6, |
| "grad_norm": 436.0, |
| "learning_rate": 6.4000000000000006e-06, |
| "loss": -97.54093017578126, |
| "step": 215 |
| }, |
| { |
| "epoch": 8.8, |
| "grad_norm": 190.0, |
| "learning_rate": 5.511111111111112e-06, |
| "loss": -96.87262573242188, |
| "step": 220 |
| }, |
| { |
| "epoch": 9.0, |
| "grad_norm": 258.0, |
| "learning_rate": 4.622222222222222e-06, |
| "loss": -95.371240234375, |
| "step": 225 |
| }, |
| { |
| "epoch": 9.2, |
| "grad_norm": 175.0, |
| "learning_rate": 3.7333333333333337e-06, |
| "loss": -95.70596923828126, |
| "step": 230 |
| }, |
| { |
| "epoch": 9.4, |
| "grad_norm": 390.0, |
| "learning_rate": 2.8444444444444446e-06, |
| "loss": -97.079931640625, |
| "step": 235 |
| }, |
| { |
| "epoch": 9.6, |
| "grad_norm": 260.0, |
| "learning_rate": 1.955555555555556e-06, |
| "loss": -97.48558349609375, |
| "step": 240 |
| }, |
| { |
| "epoch": 9.8, |
| "grad_norm": 468.0, |
| "learning_rate": 1.066666666666667e-06, |
| "loss": -98.60989379882812, |
| "step": 245 |
| }, |
| { |
| "epoch": 10.0, |
| "grad_norm": 140.0, |
| "learning_rate": 1.777777777777778e-07, |
| "loss": -98.44088134765624, |
| "step": 250 |
| }, |
| { |
| "epoch": 10.0, |
| "step": 250, |
| "total_flos": 0.0, |
| "train_loss": -70.03600473117828, |
| "train_runtime": 299.093, |
| "train_samples_per_second": 13.374, |
| "train_steps_per_second": 0.836 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 250, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 10, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": false, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|