| { |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.9986431478968792, |
| "eval_steps": 500, |
| "global_step": 138, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.03618272274988693, |
| "grad_norm": 0.0248116422444582, |
| "learning_rate": 0.0002, |
| "loss": 0.2047, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.07236544549977386, |
| "grad_norm": 0.010104305110871792, |
| "learning_rate": 0.00019930337092856243, |
| "loss": 0.0396, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.10854816824966079, |
| "grad_norm": 0.007679518777877092, |
| "learning_rate": 0.00019722318955551306, |
| "loss": 0.027, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.14473089099954772, |
| "grad_norm": 0.003990677185356617, |
| "learning_rate": 0.00019378843817721854, |
| "loss": 0.0175, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.18091361374943465, |
| "grad_norm": 0.006311236415058374, |
| "learning_rate": 0.00018904697174694447, |
| "loss": 0.017, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.21709633649932158, |
| "grad_norm": 0.004349194932729006, |
| "learning_rate": 0.0001830648511318223, |
| "loss": 0.0131, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.2532790592492085, |
| "grad_norm": 0.004049016162753105, |
| "learning_rate": 0.00017592542271443887, |
| "loss": 0.0124, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.28946178199909545, |
| "grad_norm": 0.0037118764594197273, |
| "learning_rate": 0.00016772815716257412, |
| "loss": 0.012, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.3256445047489824, |
| "grad_norm": 0.0033980372827500105, |
| "learning_rate": 0.00015858726354602248, |
| "loss": 0.0138, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.3618272274988693, |
| "grad_norm": 0.00322695542126894, |
| "learning_rate": 0.00014863009810942815, |
| "loss": 0.0136, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.39800995024875624, |
| "grad_norm": 0.004632092081010342, |
| "learning_rate": 0.000137995389871036, |
| "loss": 0.0104, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.43419267299864317, |
| "grad_norm": 0.0040463777258992195, |
| "learning_rate": 0.0001268313077693485, |
| "loss": 0.0129, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.4703753957485301, |
| "grad_norm": 0.0027721289079636335, |
| "learning_rate": 0.0001152933962873246, |
| "loss": 0.0124, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.506558118498417, |
| "grad_norm": 0.0036459509283304214, |
| "learning_rate": 0.00010354240831620541, |
| "loss": 0.0117, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.5427408412483039, |
| "grad_norm": 0.005587196908891201, |
| "learning_rate": 9.174206545276677e-05, |
| "loss": 0.0111, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.5789235639981909, |
| "grad_norm": 0.004005922935903072, |
| "learning_rate": 8.005677693484077e-05, |
| "loss": 0.0124, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.6151062867480778, |
| "grad_norm": 0.002914112526923418, |
| "learning_rate": 6.864934899622191e-05, |
| "loss": 0.0094, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.6512890094979648, |
| "grad_norm": 0.0034878209698945284, |
| "learning_rate": 5.767871655555751e-05, |
| "loss": 0.0108, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.6874717322478516, |
| "grad_norm": 0.002933693816885352, |
| "learning_rate": 4.729772884265212e-05, |
| "loss": 0.0117, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.7236544549977386, |
| "grad_norm": 0.002227836288511753, |
| "learning_rate": 3.7651019814126654e-05, |
| "loss": 0.0083, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.7598371777476255, |
| "grad_norm": 0.0033331240992993116, |
| "learning_rate": 2.8872993029040508e-05, |
| "loss": 0.0101, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.7960199004975125, |
| "grad_norm": 0.00224496191367507, |
| "learning_rate": 2.1085949060360654e-05, |
| "loss": 0.01, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.8322026232473994, |
| "grad_norm": 0.002678320510312915, |
| "learning_rate": 1.439838153227e-05, |
| "loss": 0.0085, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.8683853459972863, |
| "grad_norm": 0.001902088988572359, |
| "learning_rate": 8.903465523913957e-06, |
| "loss": 0.0112, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.9045680687471732, |
| "grad_norm": 0.003356203204020858, |
| "learning_rate": 4.6777594000230855e-06, |
| "loss": 0.0126, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.9407507914970602, |
| "grad_norm": 0.002311143558472395, |
| "learning_rate": 1.7801381552624563e-06, |
| "loss": 0.0109, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.9769335142469471, |
| "grad_norm": 0.002950450871139765, |
| "learning_rate": 2.509731335744281e-07, |
| "loss": 0.0102, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.9986431478968792, |
| "step": 138, |
| "total_flos": 1.8184278196190236e+20, |
| "train_loss": 0.02036970402991426, |
| "train_runtime": 25778.5856, |
| "train_samples_per_second": 0.343, |
| "train_steps_per_second": 0.005 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 138, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 1, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": false, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1.8184278196190236e+20, |
| "train_batch_size": 2, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|