| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 500, |
| "global_step": 2428, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.0411946446961895, |
| "grad_norm": 0.3926346004009247, |
| "learning_rate": 0.003, |
| "loss": 6.1603759765625, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.082389289392379, |
| "grad_norm": 0.42686447501182556, |
| "learning_rate": 0.003, |
| "loss": 4.739006652832031, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.12358393408856849, |
| "grad_norm": 0.5154397487640381, |
| "learning_rate": 0.003, |
| "loss": 4.014122314453125, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.164778578784758, |
| "grad_norm": 0.44306617975234985, |
| "learning_rate": 0.003, |
| "loss": 3.695447998046875, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.2059732234809475, |
| "grad_norm": 0.5973666310310364, |
| "learning_rate": 0.003, |
| "loss": 3.546534423828125, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.2059732234809475, |
| "eval_loss": 3.5068907737731934, |
| "eval_runtime": 8.733, |
| "eval_samples_per_second": 367.572, |
| "eval_steps_per_second": 46.032, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.24716786817713698, |
| "grad_norm": 0.4289214611053467, |
| "learning_rate": 0.003, |
| "loss": 3.46018798828125, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.2883625128733265, |
| "grad_norm": 0.29066139459609985, |
| "learning_rate": 0.003, |
| "loss": 3.4029681396484377, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.329557157569516, |
| "grad_norm": 0.29461681842803955, |
| "learning_rate": 0.003, |
| "loss": 3.3588226318359373, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.3707518022657055, |
| "grad_norm": 0.3961299657821655, |
| "learning_rate": 0.003, |
| "loss": 3.327505798339844, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.411946446961895, |
| "grad_norm": 0.29704517126083374, |
| "learning_rate": 0.003, |
| "loss": 3.2994342041015625, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.411946446961895, |
| "eval_loss": 3.28598952293396, |
| "eval_runtime": 8.6935, |
| "eval_samples_per_second": 369.243, |
| "eval_steps_per_second": 46.242, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.45314109165808447, |
| "grad_norm": 0.2760719358921051, |
| "learning_rate": 0.003, |
| "loss": 3.2774252319335937, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.49433573635427397, |
| "grad_norm": 0.2545988857746124, |
| "learning_rate": 0.003, |
| "loss": 3.2593310546875, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.5355303810504635, |
| "grad_norm": 0.30400964617729187, |
| "learning_rate": 0.003, |
| "loss": 3.24214599609375, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.576725025746653, |
| "grad_norm": 0.26717409491539, |
| "learning_rate": 0.003, |
| "loss": 3.2309466552734376, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.6179196704428425, |
| "grad_norm": 0.25934040546417236, |
| "learning_rate": 0.003, |
| "loss": 3.2188458251953125, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.6179196704428425, |
| "eval_loss": 3.21104097366333, |
| "eval_runtime": 8.6622, |
| "eval_samples_per_second": 370.574, |
| "eval_steps_per_second": 46.408, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.659114315139032, |
| "grad_norm": 0.2686704695224762, |
| "learning_rate": 0.003, |
| "loss": 3.2066079711914064, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.7003089598352215, |
| "grad_norm": 0.24446773529052734, |
| "learning_rate": 0.003, |
| "loss": 3.1978436279296876, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.741503604531411, |
| "grad_norm": 0.23034396767616272, |
| "learning_rate": 0.003, |
| "loss": 3.1878036499023437, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.7826982492276005, |
| "grad_norm": 0.2868192195892334, |
| "learning_rate": 0.003, |
| "loss": 3.179383544921875, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.82389289392379, |
| "grad_norm": 0.3479514420032501, |
| "learning_rate": 0.002899325557098001, |
| "loss": 3.1715652465820314, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.82389289392379, |
| "eval_loss": 3.167982816696167, |
| "eval_runtime": 9.261, |
| "eval_samples_per_second": 346.616, |
| "eval_steps_per_second": 43.408, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.8650875386199794, |
| "grad_norm": 0.21624135971069336, |
| "learning_rate": 0.0022915870802995226, |
| "loss": 3.1575164794921875, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.9062821833161689, |
| "grad_norm": 0.17490530014038086, |
| "learning_rate": 0.0013644373889485982, |
| "loss": 3.1341757202148437, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.9474768280123584, |
| "grad_norm": 0.15521171689033508, |
| "learning_rate": 0.0004919882091667163, |
| "loss": 3.109773864746094, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.9886714727085479, |
| "grad_norm": 0.13293710350990295, |
| "learning_rate": 2.6279207952956684e-05, |
| "loss": 3.0883328247070314, |
| "step": 2400 |
| }, |
| { |
| "epoch": 1.0, |
| "eval_loss": 3.0843801498413086, |
| "eval_runtime": 11.6143, |
| "eval_samples_per_second": 276.384, |
| "eval_steps_per_second": 34.613, |
| "step": 2428 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 2428, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 1, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 4331665489920000.0, |
| "train_batch_size": 400, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|