| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5, |
| "eval_steps": 50, |
| "global_step": 100, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.025, |
| "grad_norm": 0.10900866985321045, |
| "learning_rate": 1.6e-06, |
| "loss": 11.06564712524414, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.10152705758810043, |
| "learning_rate": 3.6e-06, |
| "loss": 11.040735626220703, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.075, |
| "grad_norm": 0.09875829517841339, |
| "learning_rate": 3.995627254437549e-06, |
| "loss": 11.026320648193359, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 0.09957622736692429, |
| "learning_rate": 3.9778957412029366e-06, |
| "loss": 11.012971496582031, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.125, |
| "grad_norm": 0.1074257642030716, |
| "learning_rate": 3.9466531944960116e-06, |
| "loss": 10.992045593261718, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 0.110373854637146, |
| "learning_rate": 3.902113032590307e-06, |
| "loss": 10.97699432373047, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.175, |
| "grad_norm": 0.11517436057329178, |
| "learning_rate": 3.844579509954769e-06, |
| "loss": 10.947827911376953, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 0.1078442707657814, |
| "learning_rate": 3.7744456388872234e-06, |
| "loss": 10.949494171142579, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.225, |
| "grad_norm": 0.1252269148826599, |
| "learning_rate": 3.6921905048418706e-06, |
| "loss": 10.916586303710938, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.25, |
| "grad_norm": 0.12218247354030609, |
| "learning_rate": 3.598375993789849e-06, |
| "loss": 10.902359008789062, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.25, |
| "eval_accuracy": 0.001978021978021978, |
| "eval_loss": 10.892147064208984, |
| "eval_runtime": 11.0414, |
| "eval_samples_per_second": 9.057, |
| "eval_steps_per_second": 1.54, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.275, |
| "grad_norm": 0.12842494249343872, |
| "learning_rate": 3.4936429539683075e-06, |
| "loss": 10.880451202392578, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.3, |
| "grad_norm": 0.12605500221252441, |
| "learning_rate": 3.3787068182371314e-06, |
| "loss": 10.865975952148437, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.325, |
| "grad_norm": 0.12930333614349365, |
| "learning_rate": 3.254352716947074e-06, |
| "loss": 10.846955871582031, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.35, |
| "grad_norm": 0.1300428807735443, |
| "learning_rate": 3.1214301147033453e-06, |
| "loss": 10.831236267089844, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.375, |
| "grad_norm": 0.13403981924057007, |
| "learning_rate": 2.9808470076610163e-06, |
| "loss": 10.815153503417969, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 0.1414586752653122, |
| "learning_rate": 2.833563720990581e-06, |
| "loss": 10.798695373535157, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.425, |
| "grad_norm": 0.1405540555715561, |
| "learning_rate": 2.680586348883286e-06, |
| "loss": 10.79158706665039, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.45, |
| "grad_norm": 0.15637332201004028, |
| "learning_rate": 2.5229598819076393e-06, |
| "loss": 10.773637390136718, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.475, |
| "grad_norm": 0.15481412410736084, |
| "learning_rate": 2.36176106866422e-06, |
| "loss": 10.747270965576172, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.5, |
| "grad_norm": 0.1801270991563797, |
| "learning_rate": 2.198091060500926e-06, |
| "loss": 10.725272369384765, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.5, |
| "eval_accuracy": 0.032405372405372404, |
| "eval_loss": 10.71803092956543, |
| "eval_runtime": 10.5701, |
| "eval_samples_per_second": 9.461, |
| "eval_steps_per_second": 1.608, |
| "step": 100 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 200, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 100, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 6, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|