| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 10.0, |
| "eval_steps": 500, |
| "global_step": 250, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.2, |
| "grad_norm": 19.25, |
| "learning_rate": 6.4000000000000006e-06, |
| "loss": 0.10224462747573852, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 30.625, |
| "learning_rate": 1.4400000000000001e-05, |
| "loss": 0.0848326861858368, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 32.75, |
| "learning_rate": 2.2400000000000002e-05, |
| "loss": 0.12984503507614137, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.8, |
| "grad_norm": 27.125, |
| "learning_rate": 3.0400000000000004e-05, |
| "loss": 0.10488946437835693, |
| "step": 20 |
| }, |
| { |
| "epoch": 1.0, |
| "grad_norm": 52.0, |
| "learning_rate": 3.8400000000000005e-05, |
| "loss": 0.17124745845794678, |
| "step": 25 |
| }, |
| { |
| "epoch": 1.2, |
| "grad_norm": 173.0, |
| "learning_rate": 3.9288888888888894e-05, |
| "loss": -0.6979187965393067, |
| "step": 30 |
| }, |
| { |
| "epoch": 1.4, |
| "grad_norm": 174.0, |
| "learning_rate": 3.8400000000000005e-05, |
| "loss": -0.7409286499023438, |
| "step": 35 |
| }, |
| { |
| "epoch": 1.6, |
| "grad_norm": 362.0, |
| "learning_rate": 3.7511111111111116e-05, |
| "loss": -2.4813100814819338, |
| "step": 40 |
| }, |
| { |
| "epoch": 1.8, |
| "grad_norm": 1200.0, |
| "learning_rate": 3.662222222222223e-05, |
| "loss": -35.63023376464844, |
| "step": 45 |
| }, |
| { |
| "epoch": 2.0, |
| "grad_norm": 596.0, |
| "learning_rate": 3.573333333333333e-05, |
| "loss": -81.30550537109374, |
| "step": 50 |
| }, |
| { |
| "epoch": 2.2, |
| "grad_norm": 14976.0, |
| "learning_rate": 3.4844444444444444e-05, |
| "loss": -91.82175903320312, |
| "step": 55 |
| }, |
| { |
| "epoch": 2.4, |
| "grad_norm": 2224.0, |
| "learning_rate": 3.395555555555556e-05, |
| "loss": -89.76885375976562, |
| "step": 60 |
| }, |
| { |
| "epoch": 2.6, |
| "grad_norm": 239.0, |
| "learning_rate": 3.3066666666666666e-05, |
| "loss": -90.85068359375, |
| "step": 65 |
| }, |
| { |
| "epoch": 2.8, |
| "grad_norm": 115.5, |
| "learning_rate": 3.217777777777778e-05, |
| "loss": -97.28173828125, |
| "step": 70 |
| }, |
| { |
| "epoch": 3.0, |
| "grad_norm": 114.0, |
| "learning_rate": 3.1288888888888896e-05, |
| "loss": -95.29666748046876, |
| "step": 75 |
| }, |
| { |
| "epoch": 3.2, |
| "grad_norm": 284.0, |
| "learning_rate": 3.0400000000000004e-05, |
| "loss": -92.7463623046875, |
| "step": 80 |
| }, |
| { |
| "epoch": 3.4, |
| "grad_norm": 126.5, |
| "learning_rate": 2.951111111111111e-05, |
| "loss": -96.7171142578125, |
| "step": 85 |
| }, |
| { |
| "epoch": 3.6, |
| "grad_norm": 137.0, |
| "learning_rate": 2.8622222222222223e-05, |
| "loss": -98.34115600585938, |
| "step": 90 |
| }, |
| { |
| "epoch": 3.8, |
| "grad_norm": 110.0, |
| "learning_rate": 2.7733333333333338e-05, |
| "loss": -99.39224243164062, |
| "step": 95 |
| }, |
| { |
| "epoch": 4.0, |
| "grad_norm": 108.0, |
| "learning_rate": 2.6844444444444446e-05, |
| "loss": -97.4651611328125, |
| "step": 100 |
| }, |
| { |
| "epoch": 4.2, |
| "grad_norm": 113.5, |
| "learning_rate": 2.5955555555555557e-05, |
| "loss": -99.03679809570312, |
| "step": 105 |
| }, |
| { |
| "epoch": 4.4, |
| "grad_norm": 113.0, |
| "learning_rate": 2.5066666666666672e-05, |
| "loss": -102.26163940429687, |
| "step": 110 |
| }, |
| { |
| "epoch": 4.6, |
| "grad_norm": 262.0, |
| "learning_rate": 2.417777777777778e-05, |
| "loss": -99.7196533203125, |
| "step": 115 |
| }, |
| { |
| "epoch": 4.8, |
| "grad_norm": 118.5, |
| "learning_rate": 2.328888888888889e-05, |
| "loss": -100.89335327148437, |
| "step": 120 |
| }, |
| { |
| "epoch": 5.0, |
| "grad_norm": 112.5, |
| "learning_rate": 2.2400000000000002e-05, |
| "loss": -99.00946655273438, |
| "step": 125 |
| }, |
| { |
| "epoch": 5.2, |
| "grad_norm": 133.0, |
| "learning_rate": 2.1511111111111114e-05, |
| "loss": -102.15584716796874, |
| "step": 130 |
| }, |
| { |
| "epoch": 5.4, |
| "grad_norm": 115.5, |
| "learning_rate": 2.0622222222222225e-05, |
| "loss": -104.1073486328125, |
| "step": 135 |
| }, |
| { |
| "epoch": 5.6, |
| "grad_norm": 112.5, |
| "learning_rate": 1.9733333333333336e-05, |
| "loss": -101.88866577148437, |
| "step": 140 |
| }, |
| { |
| "epoch": 5.8, |
| "grad_norm": 144.0, |
| "learning_rate": 1.8844444444444444e-05, |
| "loss": -104.56612548828124, |
| "step": 145 |
| }, |
| { |
| "epoch": 6.0, |
| "grad_norm": 182.0, |
| "learning_rate": 1.7955555555555556e-05, |
| "loss": -105.870458984375, |
| "step": 150 |
| }, |
| { |
| "epoch": 6.2, |
| "grad_norm": 118.5, |
| "learning_rate": 1.706666666666667e-05, |
| "loss": -105.58726806640625, |
| "step": 155 |
| }, |
| { |
| "epoch": 6.4, |
| "grad_norm": 114.5, |
| "learning_rate": 1.617777777777778e-05, |
| "loss": -105.295947265625, |
| "step": 160 |
| }, |
| { |
| "epoch": 6.6, |
| "grad_norm": 135.0, |
| "learning_rate": 1.528888888888889e-05, |
| "loss": -103.418505859375, |
| "step": 165 |
| }, |
| { |
| "epoch": 6.8, |
| "grad_norm": 119.5, |
| "learning_rate": 1.4400000000000001e-05, |
| "loss": -105.86998291015625, |
| "step": 170 |
| }, |
| { |
| "epoch": 7.0, |
| "grad_norm": 114.5, |
| "learning_rate": 1.3511111111111112e-05, |
| "loss": -103.21275634765625, |
| "step": 175 |
| }, |
| { |
| "epoch": 7.2, |
| "grad_norm": 118.0, |
| "learning_rate": 1.2622222222222222e-05, |
| "loss": -106.66414794921874, |
| "step": 180 |
| }, |
| { |
| "epoch": 7.4, |
| "grad_norm": 125.0, |
| "learning_rate": 1.1733333333333335e-05, |
| "loss": -104.7695556640625, |
| "step": 185 |
| }, |
| { |
| "epoch": 7.6, |
| "grad_norm": 118.0, |
| "learning_rate": 1.0844444444444446e-05, |
| "loss": -104.51566162109376, |
| "step": 190 |
| }, |
| { |
| "epoch": 7.8, |
| "grad_norm": 131.0, |
| "learning_rate": 9.955555555555556e-06, |
| "loss": -106.18931884765625, |
| "step": 195 |
| }, |
| { |
| "epoch": 8.0, |
| "grad_norm": 129.0, |
| "learning_rate": 9.066666666666667e-06, |
| "loss": -106.0660400390625, |
| "step": 200 |
| }, |
| { |
| "epoch": 8.2, |
| "grad_norm": 376.0, |
| "learning_rate": 8.177777777777779e-06, |
| "loss": -106.21219482421876, |
| "step": 205 |
| }, |
| { |
| "epoch": 8.4, |
| "grad_norm": 114.5, |
| "learning_rate": 7.28888888888889e-06, |
| "loss": -106.3694580078125, |
| "step": 210 |
| }, |
| { |
| "epoch": 8.6, |
| "grad_norm": 154.0, |
| "learning_rate": 6.4000000000000006e-06, |
| "loss": -104.23822021484375, |
| "step": 215 |
| }, |
| { |
| "epoch": 8.8, |
| "grad_norm": 120.0, |
| "learning_rate": 5.511111111111112e-06, |
| "loss": -106.70201416015625, |
| "step": 220 |
| }, |
| { |
| "epoch": 9.0, |
| "grad_norm": 111.0, |
| "learning_rate": 4.622222222222222e-06, |
| "loss": -105.63419189453126, |
| "step": 225 |
| }, |
| { |
| "epoch": 9.2, |
| "grad_norm": 127.5, |
| "learning_rate": 3.7333333333333337e-06, |
| "loss": -103.74107666015625, |
| "step": 230 |
| }, |
| { |
| "epoch": 9.4, |
| "grad_norm": 108.5, |
| "learning_rate": 2.8444444444444446e-06, |
| "loss": -105.2689697265625, |
| "step": 235 |
| }, |
| { |
| "epoch": 9.6, |
| "grad_norm": 115.0, |
| "learning_rate": 1.955555555555556e-06, |
| "loss": -106.94022216796876, |
| "step": 240 |
| }, |
| { |
| "epoch": 9.8, |
| "grad_norm": 111.0, |
| "learning_rate": 1.066666666666667e-06, |
| "loss": -107.48389892578125, |
| "step": 245 |
| }, |
| { |
| "epoch": 10.0, |
| "grad_norm": 206.0, |
| "learning_rate": 1.777777777777778e-07, |
| "loss": -107.09498291015625, |
| "step": 250 |
| }, |
| { |
| "epoch": 10.0, |
| "step": 250, |
| "total_flos": 0.0, |
| "train_loss": -84.0145669285059, |
| "train_runtime": 389.2786, |
| "train_samples_per_second": 10.275, |
| "train_steps_per_second": 0.642 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 250, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 10, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": false, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|