| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.945010183299389, |
| "eval_steps": 500, |
| "global_step": 60, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.1629327902240326, |
| "grad_norm": 2.390625, |
| "learning_rate": 1.12224987555998e-05, |
| "loss": 1.0917, |
| "mean_token_accuracy": 0.7111551932990551, |
| "num_tokens": 1976903.0, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.3258655804480652, |
| "grad_norm": 1.6015625, |
| "learning_rate": 2.525062220009955e-05, |
| "loss": 0.9665, |
| "mean_token_accuracy": 0.7346761249005794, |
| "num_tokens": 3997965.0, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.48879837067209775, |
| "grad_norm": 0.96484375, |
| "learning_rate": 3.92787456445993e-05, |
| "loss": 0.8317, |
| "mean_token_accuracy": 0.759149044007063, |
| "num_tokens": 6048529.0, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.6517311608961304, |
| "grad_norm": 0.94140625, |
| "learning_rate": 5.330686908909905e-05, |
| "loss": 0.7834, |
| "mean_token_accuracy": 0.7699160128831863, |
| "num_tokens": 8047186.0, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.814663951120163, |
| "grad_norm": 0.7890625, |
| "learning_rate": 6.73349925335988e-05, |
| "loss": 0.709, |
| "mean_token_accuracy": 0.7888295985758305, |
| "num_tokens": 9981253.0, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.9775967413441955, |
| "grad_norm": 0.734375, |
| "learning_rate": 8.136311597809856e-05, |
| "loss": 0.6825, |
| "mean_token_accuracy": 0.7937345258891583, |
| "num_tokens": 12049009.0, |
| "step": 30 |
| }, |
| { |
| "epoch": 1.1303462321792261, |
| "grad_norm": 0.8984375, |
| "learning_rate": 9.53912394225983e-05, |
| "loss": 0.6252, |
| "mean_token_accuracy": 0.80838937997818, |
| "num_tokens": 13945671.0, |
| "step": 35 |
| }, |
| { |
| "epoch": 1.2932790224032586, |
| "grad_norm": 0.74609375, |
| "learning_rate": 9.428001117497547e-05, |
| "loss": 0.5995, |
| "mean_token_accuracy": 0.8107562340795994, |
| "num_tokens": 15954175.0, |
| "step": 40 |
| }, |
| { |
| "epoch": 1.4562118126272914, |
| "grad_norm": 0.640625, |
| "learning_rate": 7.978495209059233e-05, |
| "loss": 0.5736, |
| "mean_token_accuracy": 0.8186559915542603, |
| "num_tokens": 17953792.0, |
| "step": 45 |
| }, |
| { |
| "epoch": 1.6191446028513239, |
| "grad_norm": 0.5703125, |
| "learning_rate": 5.9231925120945794e-05, |
| "loss": 0.5796, |
| "mean_token_accuracy": 0.8155237503349781, |
| "num_tokens": 19968746.0, |
| "step": 50 |
| }, |
| { |
| "epoch": 1.7820773930753564, |
| "grad_norm": 0.5546875, |
| "learning_rate": 3.938337716376685e-05, |
| "loss": 0.5553, |
| "mean_token_accuracy": 0.8232184298336506, |
| "num_tokens": 22011579.0, |
| "step": 55 |
| }, |
| { |
| "epoch": 1.945010183299389, |
| "grad_norm": 0.515625, |
| "learning_rate": 2.6769964348477097e-05, |
| "loss": 0.5419, |
| "mean_token_accuracy": 0.824396800994873, |
| "num_tokens": 24000288.0, |
| "step": 60 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 62, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 2, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 2.6377461486333542e+17, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|