| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 500, |
| "global_step": 71, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.07048458149779736, |
| "grad_norm": 0.0361328125, |
| "learning_rate": 4.971830985915493e-05, |
| "loss": 0.7936924934387207, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.14096916299559473, |
| "grad_norm": 0.0439453125, |
| "learning_rate": 4.936619718309859e-05, |
| "loss": 0.8717540740966797, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.21145374449339208, |
| "grad_norm": 0.03955078125, |
| "learning_rate": 4.9014084507042255e-05, |
| "loss": 0.8225542068481445, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.28193832599118945, |
| "grad_norm": 0.0419921875, |
| "learning_rate": 4.866197183098592e-05, |
| "loss": 0.7707944393157959, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.3524229074889868, |
| "grad_norm": 0.06298828125, |
| "learning_rate": 4.830985915492958e-05, |
| "loss": 0.8533347129821778, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.42290748898678415, |
| "grad_norm": 0.0419921875, |
| "learning_rate": 4.7957746478873244e-05, |
| "loss": 0.817015266418457, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.4933920704845815, |
| "grad_norm": 0.047607421875, |
| "learning_rate": 4.76056338028169e-05, |
| "loss": 0.8447921752929688, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.5638766519823789, |
| "grad_norm": 0.04345703125, |
| "learning_rate": 4.725352112676056e-05, |
| "loss": 0.7828725814819336, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.6343612334801763, |
| "grad_norm": 0.044677734375, |
| "learning_rate": 4.6901408450704225e-05, |
| "loss": 0.8579328536987305, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.7048458149779736, |
| "grad_norm": 0.039794921875, |
| "learning_rate": 4.654929577464789e-05, |
| "loss": 0.8333025932312011, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.775330396475771, |
| "grad_norm": 0.03955078125, |
| "learning_rate": 4.619718309859155e-05, |
| "loss": 0.8048002243041992, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.8458149779735683, |
| "grad_norm": 0.033203125, |
| "learning_rate": 4.5845070422535214e-05, |
| "loss": 0.7629288673400879, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.9162995594713657, |
| "grad_norm": 0.0322265625, |
| "learning_rate": 4.5492957746478876e-05, |
| "loss": 0.8438748359680176, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.986784140969163, |
| "grad_norm": 0.043701171875, |
| "learning_rate": 4.514084507042254e-05, |
| "loss": 0.8134084701538086, |
| "step": 70 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 710, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 10, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1.9722615323099136e+17, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|