| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.08, |
| "eval_steps": 1000, |
| "global_step": 4000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 1.0905472040176392, |
| "learning_rate": 0.0007920000000000001, |
| "loss": 7.247091674804688, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 2.1956381797790527, |
| "learning_rate": 0.001592, |
| "loss": 5.962919311523438, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 0.7685134410858154, |
| "learning_rate": 0.002392, |
| "loss": 5.536254272460938, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.6038330793380737, |
| "learning_rate": 0.003192, |
| "loss": 5.27086181640625, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 0.38015660643577576, |
| "learning_rate": 0.003992, |
| "loss": 5.034593505859375, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.25242117047309875, |
| "learning_rate": 0.004792, |
| "loss": 4.90749267578125, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 0.2624945342540741, |
| "learning_rate": 0.005592000000000001, |
| "loss": 4.80042236328125, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.17131584882736206, |
| "learning_rate": 0.006392, |
| "loss": 4.716185607910156, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 0.2284715175628662, |
| "learning_rate": 0.007192, |
| "loss": 4.674341430664063, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.24492475390434265, |
| "learning_rate": 0.007992, |
| "loss": 4.623594665527344, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_accuracy": 0.24955772994129158, |
| "eval_loss": 4.598090648651123, |
| "eval_runtime": 4.6622, |
| "eval_samples_per_second": 214.491, |
| "eval_steps_per_second": 1.716, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 0.16292935609817505, |
| "learning_rate": 0.008792, |
| "loss": 4.553927307128906, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.15358929336071014, |
| "learning_rate": 0.009592000000000002, |
| "loss": 4.5277734375, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 0.20846614241600037, |
| "learning_rate": 0.010391999999999998, |
| "loss": 4.486420288085937, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.16785487532615662, |
| "learning_rate": 0.011192, |
| "loss": 4.471618041992188, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 0.17252753674983978, |
| "learning_rate": 0.011992000000000001, |
| "loss": 4.429487915039062, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.42827746272087097, |
| "learning_rate": 0.012792, |
| "loss": 4.423707885742187, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 0.18584232032299042, |
| "learning_rate": 0.013592, |
| "loss": 4.405695190429688, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.1389407366514206, |
| "learning_rate": 0.014392, |
| "loss": 4.381403503417968, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 0.17971371114253998, |
| "learning_rate": 0.015192, |
| "loss": 4.340931091308594, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.1317117065191269, |
| "learning_rate": 0.015992, |
| "loss": 4.357503356933594, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.27022504892367905, |
| "eval_loss": 4.341521739959717, |
| "eval_runtime": 4.6212, |
| "eval_samples_per_second": 216.396, |
| "eval_steps_per_second": 1.731, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 0.3252929747104645, |
| "learning_rate": 0.016792, |
| "loss": 4.336614685058594, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.3191538155078888, |
| "learning_rate": 0.017592, |
| "loss": 4.3118289184570315, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 0.11609218269586563, |
| "learning_rate": 0.018392, |
| "loss": 4.328140563964844, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.09269429743289948, |
| "learning_rate": 0.019192, |
| "loss": 4.2995166015625, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.11243760585784912, |
| "learning_rate": 0.019992000000000003, |
| "loss": 4.306504821777343, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.1890493929386139, |
| "learning_rate": 0.019999785636239033, |
| "loss": 4.267638549804688, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 0.09589467197656631, |
| "learning_rate": 0.01999913387133122, |
| "loss": 4.277190551757813, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.21187792718410492, |
| "learning_rate": 0.019998044711915176, |
| "loss": 4.244146118164062, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 0.18817682564258575, |
| "learning_rate": 0.019996518205634257, |
| "loss": 4.229488220214844, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.1112157553434372, |
| "learning_rate": 0.0199945544192628, |
| "loss": 4.231150512695312, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.2808140900195695, |
| "eval_loss": 4.229600429534912, |
| "eval_runtime": 4.8217, |
| "eval_samples_per_second": 207.397, |
| "eval_steps_per_second": 1.659, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 0.3390547037124634, |
| "learning_rate": 0.019992153438703173, |
| "loss": 4.223145141601562, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 0.7005836367607117, |
| "learning_rate": 0.019989315368982054, |
| "loss": 4.203560791015625, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 0.1793125420808792, |
| "learning_rate": 0.019986040334245798, |
| "loss": 4.1834033203125, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 0.11384831368923187, |
| "learning_rate": 0.019982328477755038, |
| "loss": 4.1909048461914065, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 0.6594609022140503, |
| "learning_rate": 0.019978179961878402, |
| "loss": 4.169663696289063, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 0.3771114647388458, |
| "learning_rate": 0.01997359496808541, |
| "loss": 4.174803161621094, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 0.31003686785697937, |
| "learning_rate": 0.01996857369693855, |
| "loss": 4.1531503295898435, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 0.11918768286705017, |
| "learning_rate": 0.019963116368084486, |
| "loss": 4.163074951171875, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 0.5589029788970947, |
| "learning_rate": 0.01995722322024446, |
| "loss": 4.147608032226563, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 0.15616266429424286, |
| "learning_rate": 0.01995089451120385, |
| "loss": 4.122293701171875, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.287454011741683, |
| "eval_loss": 4.152503967285156, |
| "eval_runtime": 4.8042, |
| "eval_samples_per_second": 208.152, |
| "eval_steps_per_second": 1.665, |
| "step": 4000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 128, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|