Instructions to use pushpam14/api-contract-validator-grpo-7b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use pushpam14/api-contract-validator-grpo-7b with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("pushpam14/api-contract-validator-grpo-7b", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
| [ | |
| { | |
| "loss": 2.841556767663178e-09, | |
| "grad_norm": 6.955382559681311e-06, | |
| "learning_rate": 0.0, | |
| "num_tokens": 3755.0, | |
| "completions/mean_length": 86.75, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 101.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 86.75, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 101.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 86.75, | |
| "kl": 2.8415567498996097e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.0033333333333333335, | |
| "step": 1 | |
| }, | |
| { | |
| "loss": 5.243207645833081e-09, | |
| "grad_norm": 1.2843201147916261e-05, | |
| "learning_rate": 1.6666666666666668e-07, | |
| "num_tokens": 11337.0, | |
| "completions/mean_length": 93.5, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 129.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.5, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 129.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.5, | |
| "kl": 5.243207112926029e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.006666666666666667, | |
| "step": 2 | |
| }, | |
| { | |
| "loss": 1.057986764863017e-08, | |
| "grad_norm": 1.7989608750212938e-05, | |
| "learning_rate": 3.3333333333333335e-07, | |
| "num_tokens": 14991.0, | |
| "completions/mean_length": 89.5, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 94.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.5, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 94.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.5, | |
| "kl": 1.057986719388282e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.01, | |
| "step": 3 | |
| }, | |
| { | |
| "loss": 6.03428684797791e-09, | |
| "grad_norm": 1.193390016851481e-05, | |
| "learning_rate": 5.000000000000001e-07, | |
| "num_tokens": 22570.0, | |
| "completions/mean_length": 92.75, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 115.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 92.75, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 115.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 92.75, | |
| "kl": 6.034286485601115e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.013333333333333334, | |
| "step": 4 | |
| }, | |
| { | |
| "loss": 1.1485650652787172e-08, | |
| "grad_norm": 1.619913564354647e-05, | |
| "learning_rate": 6.666666666666667e-07, | |
| "num_tokens": 26194.0, | |
| "completions/mean_length": 82.0, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 86.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.0, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 86.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.0, | |
| "kl": 1.1485650134090974e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.016666666666666666, | |
| "step": 5 | |
| }, | |
| { | |
| "loss": 1.0521060467283405e-08, | |
| "grad_norm": 2.1455727619468234e-05, | |
| "learning_rate": 8.333333333333333e-07, | |
| "num_tokens": 29760.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 57.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 57.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.5, | |
| "kl": 1.0521059436996438e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.02, | |
| "step": 6 | |
| }, | |
| { | |
| "loss": 5.12293052423729e-09, | |
| "grad_norm": 2.5380790248163976e-05, | |
| "learning_rate": 1.0000000000000002e-06, | |
| "num_tokens": 33878.0, | |
| "completions/mean_length": 62.5, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 64.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 62.5, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 64.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 62.5, | |
| "kl": 5.122929678691435e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.023333333333333334, | |
| "step": 7 | |
| }, | |
| { | |
| "loss": 6.705522537231445e-08, | |
| "grad_norm": 0.9678249955177307, | |
| "learning_rate": 1.1666666666666668e-06, | |
| "num_tokens": 37519.0, | |
| "completions/mean_length": 86.25, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 106.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 86.25, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 106.0, | |
| "rewards/reward_fn/mean": 2.0250000953674316, | |
| "rewards/reward_fn/std": 0.7500000596046448, | |
| "reward": 2.0250000953674316, | |
| "reward_std": 0.7500000596046448, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 86.25, | |
| "kl": 8.57097404605156e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.02666666666666667, | |
| "step": 8 | |
| }, | |
| { | |
| "loss": 1.0378420789436404e-08, | |
| "grad_norm": 2.000186032091733e-05, | |
| "learning_rate": 1.3333333333333334e-06, | |
| "num_tokens": 40598.0, | |
| "completions/mean_length": 69.75, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 69.75, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 69.75, | |
| "kl": 1.0378420540746447e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.03, | |
| "step": 9 | |
| }, | |
| { | |
| "loss": 0.0, | |
| "grad_norm": 1.1519451141357422, | |
| "learning_rate": 1.5e-06, | |
| "num_tokens": 47147.0, | |
| "completions/mean_length": 80.25, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.25, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": -0.15000000596046448, | |
| "rewards/reward_fn/std": 0.30000004172325134, | |
| "reward": -0.15000000596046448, | |
| "reward_std": 0.30000001192092896, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 80.25, | |
| "kl": 1.2190834127068229e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.03333333333333333, | |
| "step": 10 | |
| }, | |
| { | |
| "loss": 1.133202598424532e-08, | |
| "grad_norm": 1.586303005751688e-05, | |
| "learning_rate": 1.6666666666666667e-06, | |
| "num_tokens": 50765.0, | |
| "completions/mean_length": 80.5, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 84.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.5, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 84.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.5, | |
| "kl": 1.1332024939747498e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.03666666666666667, | |
| "step": 11 | |
| }, | |
| { | |
| "loss": 2.3189201669993054e-08, | |
| "grad_norm": 7.569757872261107e-05, | |
| "learning_rate": 1.8333333333333333e-06, | |
| "num_tokens": 54882.0, | |
| "completions/mean_length": 60.25, | |
| "completions/min_length": 52.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 60.25, | |
| "completions/min_terminated_length": 52.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 60.25, | |
| "kl": 2.3189201101558865e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.04, | |
| "step": 12 | |
| }, | |
| { | |
| "loss": 3.0306799292389996e-09, | |
| "grad_norm": 8.378610800718889e-06, | |
| "learning_rate": 2.0000000000000003e-06, | |
| "num_tokens": 60252.0, | |
| "completions/mean_length": 77.5, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 101.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 77.5, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 101.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 77.5, | |
| "kl": 3.030679636140121e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.043333333333333335, | |
| "step": 13 | |
| }, | |
| { | |
| "loss": 4.7533941227584364e-09, | |
| "grad_norm": 8.288271601486485e-06, | |
| "learning_rate": 2.166666666666667e-06, | |
| "num_tokens": 64002.0, | |
| "completions/mean_length": 85.5, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 93.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 85.5, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 93.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 85.5, | |
| "kl": 4.75339371064365e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.04666666666666667, | |
| "step": 14 | |
| }, | |
| { | |
| "loss": -2.9802322387695312e-08, | |
| "grad_norm": 1.5125610828399658, | |
| "learning_rate": 2.3333333333333336e-06, | |
| "num_tokens": 67284.0, | |
| "completions/mean_length": 95.5, | |
| "completions/min_length": 86.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 95.5, | |
| "completions/min_terminated_length": 86.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 0.75, | |
| "rewards/reward_fn/std": 2.5, | |
| "reward": 0.75, | |
| "reward_std": 2.5, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 95.5, | |
| "kl": 6.797420667226106e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.05, | |
| "step": 15 | |
| }, | |
| { | |
| "loss": 4.780138507243237e-09, | |
| "grad_norm": 5.973496627120767e-06, | |
| "learning_rate": 2.5e-06, | |
| "num_tokens": 74891.0, | |
| "completions/mean_length": 99.75, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 127.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 99.75, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 127.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 99.75, | |
| "kl": 4.780138340265694e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.05333333333333334, | |
| "step": 16 | |
| }, | |
| { | |
| "loss": 5.21540641784668e-08, | |
| "grad_norm": 1.0008879899978638, | |
| "learning_rate": 2.666666666666667e-06, | |
| "num_tokens": 81444.0, | |
| "completions/mean_length": 81.25, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 81.25, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": -0.15000000596046448, | |
| "rewards/reward_fn/std": 0.30000004172325134, | |
| "reward": -0.15000000596046448, | |
| "reward_std": 0.30000001192092896, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 81.25, | |
| "kl": 6.453393143601716e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.056666666666666664, | |
| "step": 17 | |
| }, | |
| { | |
| "loss": 3.6009386583657488e-09, | |
| "grad_norm": 7.384104264929192e-06, | |
| "learning_rate": 2.8333333333333335e-06, | |
| "num_tokens": 85202.0, | |
| "completions/mean_length": 87.5, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 96.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.5, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 96.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.5, | |
| "kl": 3.6009383208579493e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.06, | |
| "step": 18 | |
| }, | |
| { | |
| "loss": 7.772978705133937e-09, | |
| "grad_norm": 1.535337469249498e-05, | |
| "learning_rate": 3e-06, | |
| "num_tokens": 88282.0, | |
| "completions/mean_length": 70.0, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.0, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.0, | |
| "kl": 7.772977426157013e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.06333333333333334, | |
| "step": 19 | |
| }, | |
| { | |
| "loss": 9.086053687212825e-09, | |
| "grad_norm": 1.2720918675768189e-05, | |
| "learning_rate": 3.1666666666666667e-06, | |
| "num_tokens": 91363.0, | |
| "completions/mean_length": 70.25, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.25, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.25, | |
| "kl": 9.08605329641432e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.06666666666666667, | |
| "step": 20 | |
| }, | |
| { | |
| "loss": 9.313173343628023e-09, | |
| "grad_norm": 1.242417965841014e-05, | |
| "learning_rate": 3.3333333333333333e-06, | |
| "num_tokens": 95122.0, | |
| "completions/mean_length": 87.75, | |
| "completions/min_length": 86.0, | |
| "completions/max_length": 90.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.75, | |
| "completions/min_terminated_length": 86.0, | |
| "completions/max_terminated_length": 90.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.75, | |
| "kl": 9.313172881775245e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.07, | |
| "step": 21 | |
| }, | |
| { | |
| "loss": 8.766677694893588e-08, | |
| "grad_norm": 8.498241368215531e-05, | |
| "learning_rate": 3.5e-06, | |
| "num_tokens": 98744.0, | |
| "completions/mean_length": 81.5, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 81.5, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 81.5, | |
| "kl": 8.766677137828083e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.07333333333333333, | |
| "step": 22 | |
| }, | |
| { | |
| "loss": 3.522295122238006e-09, | |
| "grad_norm": 7.00853297530557e-06, | |
| "learning_rate": 3.6666666666666666e-06, | |
| "num_tokens": 102482.0, | |
| "completions/mean_length": 82.5, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.5, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.5, | |
| "kl": 3.522294903746115e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.07666666666666666, | |
| "step": 23 | |
| }, | |
| { | |
| "loss": 1.1255299803281105e-08, | |
| "grad_norm": 1.763108593877405e-05, | |
| "learning_rate": 3.833333333333334e-06, | |
| "num_tokens": 106249.0, | |
| "completions/mean_length": 89.75, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 107.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.75, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 107.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.75, | |
| "kl": 1.1255299909862515e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.08, | |
| "step": 24 | |
| }, | |
| { | |
| "loss": 7.503942356379412e-09, | |
| "grad_norm": 1.5865898603806272e-05, | |
| "learning_rate": 4.000000000000001e-06, | |
| "num_tokens": 109301.0, | |
| "completions/mean_length": 63.0, | |
| "completions/min_length": 59.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 63.0, | |
| "completions/min_terminated_length": 59.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 63.0, | |
| "kl": 7.503943095343857e-06, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.08333333333333333, | |
| "step": 25 | |
| }, | |
| { | |
| "loss": -1.1920928955078125e-07, | |
| "grad_norm": 0.6811502575874329, | |
| "learning_rate": 4.166666666666667e-06, | |
| "num_tokens": 116944.0, | |
| "completions/mean_length": 108.75, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 169.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 108.75, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 169.0, | |
| "rewards/reward_fn/mean": 0.6499999761581421, | |
| "rewards/reward_fn/std": 0.40414518117904663, | |
| "reward": 0.6499999761581421, | |
| "reward_std": 0.40414518117904663, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 108.75, | |
| "kl": 1.609930473023269e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.08666666666666667, | |
| "step": 26 | |
| }, | |
| { | |
| "loss": 6.183981895446777e-07, | |
| "grad_norm": 0.8662334084510803, | |
| "learning_rate": 4.333333333333334e-06, | |
| "num_tokens": 120224.0, | |
| "completions/mean_length": 95.0, | |
| "completions/min_length": 90.0, | |
| "completions/max_length": 99.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 95.0, | |
| "completions/min_terminated_length": 90.0, | |
| "completions/max_terminated_length": 99.0, | |
| "rewards/reward_fn/mean": 0.75, | |
| "rewards/reward_fn/std": 2.5, | |
| "reward": 0.75, | |
| "reward_std": 2.5, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 95.0, | |
| "kl": 0.000639034085907042, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.09, | |
| "step": 27 | |
| }, | |
| { | |
| "loss": 6.729024448759446e-08, | |
| "grad_norm": 0.00011963916040258482, | |
| "learning_rate": 4.5e-06, | |
| "num_tokens": 124340.0, | |
| "completions/mean_length": 70.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 76.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 76.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.0, | |
| "kl": 6.72902469887049e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.09333333333333334, | |
| "step": 28 | |
| }, | |
| { | |
| "loss": 5.487968124384679e-08, | |
| "grad_norm": 0.00010476292663952336, | |
| "learning_rate": 4.666666666666667e-06, | |
| "num_tokens": 129693.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 54.0, | |
| "completions/max_length": 95.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 54.0, | |
| "completions/max_terminated_length": 95.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 5.487967632689106e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.09666666666666666, | |
| "step": 29 | |
| }, | |
| { | |
| "loss": 1.517949357321413e-07, | |
| "grad_norm": 7.261318387463689e-05, | |
| "learning_rate": 4.833333333333333e-06, | |
| "num_tokens": 133362.0, | |
| "completions/mean_length": 93.25, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 106.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.25, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 106.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.25, | |
| "kl": 0.00015179493675532285, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.1, | |
| "step": 30 | |
| }, | |
| { | |
| "loss": 3.940981940786514e-08, | |
| "grad_norm": 3.8212026993278414e-05, | |
| "learning_rate": 5e-06, | |
| "num_tokens": 137470.0, | |
| "completions/mean_length": 59.0, | |
| "completions/min_length": 50.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 59.0, | |
| "completions/min_terminated_length": 50.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 59.0, | |
| "kl": 3.940982151107164e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.10333333333333333, | |
| "step": 31 | |
| }, | |
| { | |
| "loss": 2.1316813558769354e-07, | |
| "grad_norm": 9.725688141770661e-05, | |
| "learning_rate": 4.981481481481482e-06, | |
| "num_tokens": 141099.0, | |
| "completions/mean_length": 83.25, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 94.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.25, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 94.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.25, | |
| "kl": 0.0002131681339960778, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.10666666666666667, | |
| "step": 32 | |
| }, | |
| { | |
| "loss": 1.759407730617113e-08, | |
| "grad_norm": 1.6773901734268293e-05, | |
| "learning_rate": 4.962962962962964e-06, | |
| "num_tokens": 144839.0, | |
| "completions/mean_length": 83.0, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.0, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.0, | |
| "kl": 1.7594075870874804e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.11, | |
| "step": 33 | |
| }, | |
| { | |
| "loss": 5.759298801422119e-06, | |
| "grad_norm": 1.037492036819458, | |
| "learning_rate": 4.944444444444445e-06, | |
| "num_tokens": 151345.0, | |
| "completions/mean_length": 69.5, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 69.5, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": -0.15000000596046448, | |
| "rewards/reward_fn/std": 0.30000004172325134, | |
| "reward": -0.15000000596046448, | |
| "reward_std": 0.30000001192092896, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 69.5, | |
| "kl": 0.005793202784843743, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.11333333333333333, | |
| "step": 34 | |
| }, | |
| { | |
| "loss": 6.649032684435952e-07, | |
| "grad_norm": 0.0002313987206434831, | |
| "learning_rate": 4.925925925925926e-06, | |
| "num_tokens": 154616.0, | |
| "completions/mean_length": 92.75, | |
| "completions/min_length": 83.0, | |
| "completions/max_length": 98.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 92.75, | |
| "completions/min_terminated_length": 83.0, | |
| "completions/max_terminated_length": 98.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 92.75, | |
| "kl": 0.0006649032875429839, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.11666666666666667, | |
| "step": 35 | |
| }, | |
| { | |
| "loss": 7.464942086699011e-08, | |
| "grad_norm": 5.705406147171743e-05, | |
| "learning_rate": 4.907407407407408e-06, | |
| "num_tokens": 162227.0, | |
| "completions/mean_length": 100.75, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 120.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 100.75, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 120.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 100.75, | |
| "kl": 7.464942427759524e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.12, | |
| "step": 36 | |
| }, | |
| { | |
| "loss": 1.0281801223754883e-05, | |
| "grad_norm": 0.8818839192390442, | |
| "learning_rate": 4.888888888888889e-06, | |
| "num_tokens": 168774.0, | |
| "completions/mean_length": 79.75, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.75, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 0.0, | |
| "rewards/reward_fn/std": 0.3464101552963257, | |
| "reward": 0.0, | |
| "reward_std": 0.3464101552963257, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 79.75, | |
| "kl": 0.01026601018384099, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.12333333333333334, | |
| "step": 37 | |
| }, | |
| { | |
| "loss": 5.327683538780548e-07, | |
| "grad_norm": 0.00019790360238403082, | |
| "learning_rate": 4.870370370370371e-06, | |
| "num_tokens": 172446.0, | |
| "completions/mean_length": 94.0, | |
| "completions/min_length": 88.0, | |
| "completions/max_length": 99.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 94.0, | |
| "completions/min_terminated_length": 88.0, | |
| "completions/max_terminated_length": 99.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 94.0, | |
| "kl": 0.0005327683211362455, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.12666666666666668, | |
| "step": 38 | |
| }, | |
| { | |
| "loss": 3.485862265506512e-08, | |
| "grad_norm": 4.8746464017312974e-05, | |
| "learning_rate": 4.851851851851852e-06, | |
| "num_tokens": 175541.0, | |
| "completions/mean_length": 73.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.75, | |
| "kl": 3.485861861918238e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.13, | |
| "step": 39 | |
| }, | |
| { | |
| "loss": 1.1753998478525318e-05, | |
| "grad_norm": 0.00042901738197542727, | |
| "learning_rate": 4.833333333333333e-06, | |
| "num_tokens": 182083.0, | |
| "completions/mean_length": 78.5, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 78.5, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 0.30000001192092896, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 0.30000001192092896, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 78.5, | |
| "kl": 0.011753997765481472, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.13333333333333333, | |
| "step": 40 | |
| }, | |
| { | |
| "loss": 2.8200943802403344e-07, | |
| "grad_norm": 9.72169145825319e-05, | |
| "learning_rate": 4.814814814814815e-06, | |
| "num_tokens": 187469.0, | |
| "completions/mean_length": 81.5, | |
| "completions/min_length": 68.0, | |
| "completions/max_length": 103.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 81.5, | |
| "completions/min_terminated_length": 68.0, | |
| "completions/max_terminated_length": 103.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 81.5, | |
| "kl": 0.0002820094414346386, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.13666666666666666, | |
| "step": 41 | |
| }, | |
| { | |
| "loss": 4.7663434088462964e-07, | |
| "grad_norm": 0.0003350832557771355, | |
| "learning_rate": 4.796296296296297e-06, | |
| "num_tokens": 193338.0, | |
| "completions/mean_length": 82.25, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.25, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.25, | |
| "kl": 0.000476634315418778, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.14, | |
| "step": 42 | |
| }, | |
| { | |
| "loss": 2.04133328907119e-07, | |
| "grad_norm": 6.243437383091077e-05, | |
| "learning_rate": 4.777777777777778e-06, | |
| "num_tokens": 199243.0, | |
| "completions/mean_length": 91.25, | |
| "completions/min_length": 84.0, | |
| "completions/max_length": 99.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 91.25, | |
| "completions/min_terminated_length": 84.0, | |
| "completions/max_terminated_length": 99.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 91.25, | |
| "kl": 0.00020413331822055625, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.14333333333333334, | |
| "step": 43 | |
| }, | |
| { | |
| "loss": 1.2609015698217263e-07, | |
| "grad_norm": 3.512680996209383e-05, | |
| "learning_rate": 4.75925925925926e-06, | |
| "num_tokens": 206854.0, | |
| "completions/mean_length": 100.75, | |
| "completions/min_length": 86.0, | |
| "completions/max_length": 122.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 100.75, | |
| "completions/min_terminated_length": 86.0, | |
| "completions/max_terminated_length": 122.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 100.75, | |
| "kl": 0.0001260901499335887, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.14666666666666667, | |
| "step": 44 | |
| }, | |
| { | |
| "loss": 2.457550749568327e-07, | |
| "grad_norm": 9.453335223952308e-05, | |
| "learning_rate": 4.7407407407407415e-06, | |
| "num_tokens": 210458.0, | |
| "completions/mean_length": 77.0, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 77.0, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 77.0, | |
| "kl": 0.00024575508359703235, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.15, | |
| "step": 45 | |
| }, | |
| { | |
| "loss": 4.019221933049266e-07, | |
| "grad_norm": 0.0001815123250707984, | |
| "learning_rate": 4.722222222222222e-06, | |
| "num_tokens": 216329.0, | |
| "completions/mean_length": 82.75, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 87.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.75, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 87.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.75, | |
| "kl": 0.0004019221851194743, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.15333333333333332, | |
| "step": 46 | |
| }, | |
| { | |
| "loss": 1.3853189102519536e-06, | |
| "grad_norm": 0.0004176298971287906, | |
| "learning_rate": 4.703703703703704e-06, | |
| "num_tokens": 219603.0, | |
| "completions/mean_length": 93.5, | |
| "completions/min_length": 84.0, | |
| "completions/max_length": 109.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.5, | |
| "completions/min_terminated_length": 84.0, | |
| "completions/max_terminated_length": 109.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.5, | |
| "kl": 0.0013853188283974305, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.15666666666666668, | |
| "step": 47 | |
| }, | |
| { | |
| "loss": 1.2071354831277858e-05, | |
| "grad_norm": 0.0002438570372760296, | |
| "learning_rate": 4.6851851851851855e-06, | |
| "num_tokens": 226095.0, | |
| "completions/mean_length": 66.0, | |
| "completions/min_length": 58.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.0, | |
| "completions/min_terminated_length": 58.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 0.30000001192092896, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 0.30000001192092896, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.0, | |
| "kl": 0.01207135320873931, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.16, | |
| "step": 48 | |
| }, | |
| { | |
| "loss": 1.8700957298278809e-06, | |
| "grad_norm": 1.0794061422348022, | |
| "learning_rate": 4.666666666666667e-06, | |
| "num_tokens": 229355.0, | |
| "completions/mean_length": 90.0, | |
| "completions/min_length": 84.0, | |
| "completions/max_length": 98.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.0, | |
| "completions/min_terminated_length": 84.0, | |
| "completions/max_terminated_length": 98.0, | |
| "rewards/reward_fn/mean": 0.75, | |
| "rewards/reward_fn/std": 2.5, | |
| "reward": 0.75, | |
| "reward_std": 2.5, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 90.0, | |
| "kl": 0.0018941380403703079, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.16333333333333333, | |
| "step": 49 | |
| }, | |
| { | |
| "loss": 0.00011649727821350098, | |
| "grad_norm": 1.1730598211288452, | |
| "learning_rate": 4.648148148148148e-06, | |
| "num_tokens": 235903.0, | |
| "completions/mean_length": 80.0, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.0, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 0.07750000059604645, | |
| "rewards/reward_fn/std": 0.28640007972717285, | |
| "reward": 0.07750000059604645, | |
| "reward_std": 0.28640007972717285, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 80.0, | |
| "kl": 0.11645952356047928, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.16666666666666666, | |
| "step": 50 | |
| }, | |
| { | |
| "loss": 1.6100704669952393e-05, | |
| "grad_norm": 1.0840357542037964, | |
| "learning_rate": 4.62962962962963e-06, | |
| "num_tokens": 242424.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 68.0, | |
| "completions/max_length": 76.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 68.0, | |
| "completions/max_terminated_length": 76.0, | |
| "rewards/reward_fn/mean": 0.4749999940395355, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.4749999940395355, | |
| "reward_std": 0.34999996423721313, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 73.25, | |
| "kl": 0.01610480691306293, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.17, | |
| "step": 51 | |
| }, | |
| { | |
| "loss": 4.018417598672386e-07, | |
| "grad_norm": 0.00022893736604601145, | |
| "learning_rate": 4.611111111111112e-06, | |
| "num_tokens": 246513.0, | |
| "completions/mean_length": 58.25, | |
| "completions/min_length": 51.0, | |
| "completions/max_length": 69.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 58.25, | |
| "completions/min_terminated_length": 51.0, | |
| "completions/max_terminated_length": 69.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 58.25, | |
| "kl": 0.00040184172394219786, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.17333333333333334, | |
| "step": 52 | |
| }, | |
| { | |
| "loss": 6.96572499236936e-07, | |
| "grad_norm": 0.00021773447224404663, | |
| "learning_rate": 4.592592592592593e-06, | |
| "num_tokens": 250174.0, | |
| "completions/mean_length": 91.25, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 109.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 91.25, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 109.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 91.25, | |
| "kl": 0.0006965724314795807, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.17666666666666667, | |
| "step": 53 | |
| }, | |
| { | |
| "loss": 3.4793967529367364e-07, | |
| "grad_norm": 0.00016281715943478048, | |
| "learning_rate": 4.5740740740740745e-06, | |
| "num_tokens": 257762.0, | |
| "completions/mean_length": 95.0, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 113.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 95.0, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 113.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 95.0, | |
| "kl": 0.0003479396655166056, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.18, | |
| "step": 54 | |
| }, | |
| { | |
| "loss": 4.616466355855664e-07, | |
| "grad_norm": 0.00012537445581983775, | |
| "learning_rate": 4.555555555555556e-06, | |
| "num_tokens": 261909.0, | |
| "completions/mean_length": 65.75, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.75, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.75, | |
| "kl": 0.0004616465848812368, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.18333333333333332, | |
| "step": 55 | |
| }, | |
| { | |
| "loss": 2.5688721052574692e-06, | |
| "grad_norm": 0.0007535706390626729, | |
| "learning_rate": 4.537037037037038e-06, | |
| "num_tokens": 269508.0, | |
| "completions/mean_length": 97.75, | |
| "completions/min_length": 84.0, | |
| "completions/max_length": 107.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 97.75, | |
| "completions/min_terminated_length": 84.0, | |
| "completions/max_terminated_length": 107.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 97.75, | |
| "kl": 0.0025688721070764586, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.18666666666666668, | |
| "step": 56 | |
| }, | |
| { | |
| "loss": 6.940192065485462e-07, | |
| "grad_norm": 8.615383558208123e-05, | |
| "learning_rate": 4.5185185185185185e-06, | |
| "num_tokens": 272835.0, | |
| "completions/mean_length": 106.75, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 140.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 106.75, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 140.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 106.75, | |
| "kl": 0.0006940192106412724, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.19, | |
| "step": 57 | |
| }, | |
| { | |
| "loss": 7.025802233329159e-07, | |
| "grad_norm": 0.00026685651391744614, | |
| "learning_rate": 4.5e-06, | |
| "num_tokens": 278186.0, | |
| "completions/mean_length": 72.75, | |
| "completions/min_length": 55.0, | |
| "completions/max_length": 98.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.75, | |
| "completions/min_terminated_length": 55.0, | |
| "completions/max_terminated_length": 98.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.75, | |
| "kl": 0.0007025802042335272, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.19333333333333333, | |
| "step": 58 | |
| }, | |
| { | |
| "loss": 7.099001777532976e-07, | |
| "grad_norm": 0.0002469788014423102, | |
| "learning_rate": 4.481481481481482e-06, | |
| "num_tokens": 284058.0, | |
| "completions/mean_length": 83.0, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 95.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.0, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 95.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.0, | |
| "kl": 0.0007099001741153188, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.19666666666666666, | |
| "step": 59 | |
| }, | |
| { | |
| "loss": 8.496735404150968e-07, | |
| "grad_norm": 0.0003938635636586696, | |
| "learning_rate": 4.462962962962963e-06, | |
| "num_tokens": 288146.0, | |
| "completions/mean_length": 60.0, | |
| "completions/min_length": 49.0, | |
| "completions/max_length": 66.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 60.0, | |
| "completions/min_terminated_length": 49.0, | |
| "completions/max_terminated_length": 66.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 60.0, | |
| "kl": 0.0008496734662912786, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.2, | |
| "step": 60 | |
| }, | |
| { | |
| "loss": 6.281337050495495e-07, | |
| "grad_norm": 0.000125624515931122, | |
| "learning_rate": 4.444444444444444e-06, | |
| "num_tokens": 294056.0, | |
| "completions/mean_length": 92.5, | |
| "completions/min_length": 83.0, | |
| "completions/max_length": 99.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 92.5, | |
| "completions/min_terminated_length": 83.0, | |
| "completions/max_terminated_length": 99.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 92.5, | |
| "kl": 0.000628133631835226, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.20333333333333334, | |
| "step": 61 | |
| }, | |
| { | |
| "loss": 8.822756569770718e-08, | |
| "grad_norm": 6.231183942873031e-05, | |
| "learning_rate": 4.425925925925927e-06, | |
| "num_tokens": 297108.0, | |
| "completions/mean_length": 63.0, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 66.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 63.0, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 66.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 63.0, | |
| "kl": 8.822756444715196e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.20666666666666667, | |
| "step": 62 | |
| }, | |
| { | |
| "loss": 1.6197562217712402e-05, | |
| "grad_norm": 0.921194314956665, | |
| "learning_rate": 4.407407407407408e-06, | |
| "num_tokens": 303629.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 0.0, | |
| "rewards/reward_fn/std": 0.3464101552963257, | |
| "reward": 0.0, | |
| "reward_std": 0.3464101552963257, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 73.25, | |
| "kl": 0.016220400109887123, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.21, | |
| "step": 63 | |
| }, | |
| { | |
| "loss": 5.1808804357733607e-08, | |
| "grad_norm": 2.911170304287225e-05, | |
| "learning_rate": 4.388888888888889e-06, | |
| "num_tokens": 306704.0, | |
| "completions/mean_length": 68.75, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.75, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 68.75, | |
| "kl": 5.180879816180095e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.21333333333333335, | |
| "step": 64 | |
| }, | |
| { | |
| "loss": 1.0601124813547358e-06, | |
| "grad_norm": 0.00015854882076382637, | |
| "learning_rate": 4.370370370370371e-06, | |
| "num_tokens": 309991.0, | |
| "completions/mean_length": 96.75, | |
| "completions/min_length": 88.0, | |
| "completions/max_length": 110.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.75, | |
| "completions/min_terminated_length": 88.0, | |
| "completions/max_terminated_length": 110.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.75, | |
| "kl": 0.0010601124668028206, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.21666666666666667, | |
| "step": 65 | |
| }, | |
| { | |
| "loss": 1.7881393432617188e-05, | |
| "grad_norm": 0.9491147398948669, | |
| "learning_rate": 4.351851851851852e-06, | |
| "num_tokens": 316540.0, | |
| "completions/mean_length": 80.25, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.25, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 0.15000000596046448, | |
| "rewards/reward_fn/std": 0.30000004172325134, | |
| "reward": 0.15000000596046448, | |
| "reward_std": 0.30000001192092896, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 80.25, | |
| "kl": 0.017915051896125078, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.22, | |
| "step": 66 | |
| }, | |
| { | |
| "loss": 1.6540288925170898e-06, | |
| "grad_norm": 1.4446711540222168, | |
| "learning_rate": 4.333333333333334e-06, | |
| "num_tokens": 320655.0, | |
| "completions/mean_length": 58.75, | |
| "completions/min_length": 55.0, | |
| "completions/max_length": 62.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 58.75, | |
| "completions/min_terminated_length": 55.0, | |
| "completions/max_terminated_length": 62.0, | |
| "rewards/reward_fn/mean": 0.6499999761581421, | |
| "rewards/reward_fn/std": 0.40414518117904663, | |
| "reward": 0.6499999761581421, | |
| "reward_std": 0.40414518117904663, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 58.75, | |
| "kl": 0.0017991842469200492, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.22333333333333333, | |
| "step": 67 | |
| }, | |
| { | |
| "loss": 5.843520511916722e-07, | |
| "grad_norm": 0.00017113692592829466, | |
| "learning_rate": 4.314814814814815e-06, | |
| "num_tokens": 324291.0, | |
| "completions/mean_length": 85.0, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 96.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 85.0, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 96.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 85.0, | |
| "kl": 0.0005843520339112729, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.22666666666666666, | |
| "step": 68 | |
| }, | |
| { | |
| "loss": 2.271695166200516e-06, | |
| "grad_norm": 0.0006372662610374391, | |
| "learning_rate": 4.296296296296296e-06, | |
| "num_tokens": 330151.0, | |
| "completions/mean_length": 80.0, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 97.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.0, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 97.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.0, | |
| "kl": 0.0022716949169989675, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.23, | |
| "step": 69 | |
| }, | |
| { | |
| "loss": 1.3857891190127702e-06, | |
| "grad_norm": 0.0004074950120411813, | |
| "learning_rate": 4.277777777777778e-06, | |
| "num_tokens": 336039.0, | |
| "completions/mean_length": 87.0, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 97.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.0, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 97.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.0, | |
| "kl": 0.0013857891026418656, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.23333333333333334, | |
| "step": 70 | |
| }, | |
| { | |
| "loss": 4.3999165200148127e-07, | |
| "grad_norm": 0.0001410327386111021, | |
| "learning_rate": 4.2592592592592596e-06, | |
| "num_tokens": 339694.0, | |
| "completions/mean_length": 89.75, | |
| "completions/min_length": 81.0, | |
| "completions/max_length": 103.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.75, | |
| "completions/min_terminated_length": 81.0, | |
| "completions/max_terminated_length": 103.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.75, | |
| "kl": 0.0004399916433612816, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.23666666666666666, | |
| "step": 71 | |
| }, | |
| { | |
| "loss": 4.500150680541992e-06, | |
| "grad_norm": 0.9828253984451294, | |
| "learning_rate": 4.240740740740741e-06, | |
| "num_tokens": 343800.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 59.0, | |
| "completions/max_length": 79.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 59.0, | |
| "completions/max_terminated_length": 79.0, | |
| "rewards/reward_fn/mean": 0.4749999940395355, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.4749999940395355, | |
| "reward_std": 0.34999996423721313, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 67.5, | |
| "kl": 0.004503751988522708, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.24, | |
| "step": 72 | |
| }, | |
| { | |
| "loss": 9.226000656781252e-07, | |
| "grad_norm": 0.00012342410627752542, | |
| "learning_rate": 4.222222222222223e-06, | |
| "num_tokens": 351429.0, | |
| "completions/mean_length": 105.25, | |
| "completions/min_length": 83.0, | |
| "completions/max_length": 123.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 105.25, | |
| "completions/min_terminated_length": 83.0, | |
| "completions/max_terminated_length": 123.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 105.25, | |
| "kl": 0.0009225999674526975, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.24333333333333335, | |
| "step": 73 | |
| }, | |
| { | |
| "loss": 1.010352320918173e-06, | |
| "grad_norm": 0.0002404086699243635, | |
| "learning_rate": 4.2037037037037045e-06, | |
| "num_tokens": 356833.0, | |
| "completions/mean_length": 86.0, | |
| "completions/min_length": 70.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 86.0, | |
| "completions/min_terminated_length": 70.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 86.0, | |
| "kl": 0.0010103522799909115, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.24666666666666667, | |
| "step": 74 | |
| }, | |
| { | |
| "loss": 5.172425403543457e-07, | |
| "grad_norm": 7.805436325725168e-05, | |
| "learning_rate": 4.185185185185185e-06, | |
| "num_tokens": 364431.0, | |
| "completions/mean_length": 97.5, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 121.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 97.5, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 121.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 97.5, | |
| "kl": 0.0005172425153432414, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.25, | |
| "step": 75 | |
| }, | |
| { | |
| "loss": 2.808958015521057e-06, | |
| "grad_norm": 0.00032745281350798905, | |
| "learning_rate": 4.166666666666667e-06, | |
| "num_tokens": 368516.0, | |
| "completions/mean_length": 64.25, | |
| "completions/min_length": 57.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.25, | |
| "completions/min_terminated_length": 57.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 0.30000001192092896, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 0.30000001192092896, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.25, | |
| "kl": 0.002808958204695955, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.25333333333333335, | |
| "step": 76 | |
| }, | |
| { | |
| "loss": 2.085314190480858e-05, | |
| "grad_norm": 0.0047827353700995445, | |
| "learning_rate": 4.1481481481481485e-06, | |
| "num_tokens": 372606.0, | |
| "completions/mean_length": 66.5, | |
| "completions/min_length": 58.0, | |
| "completions/max_length": 74.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.5, | |
| "completions/min_terminated_length": 58.0, | |
| "completions/max_terminated_length": 74.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.5, | |
| "kl": 0.02085313602583483, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.25666666666666665, | |
| "step": 77 | |
| }, | |
| { | |
| "loss": 1.7112019179421623e-07, | |
| "grad_norm": 8.766031533014029e-05, | |
| "learning_rate": 4.12962962962963e-06, | |
| "num_tokens": 375689.0, | |
| "completions/mean_length": 70.75, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.75, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.75, | |
| "kl": 0.00017112019122578204, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.26, | |
| "step": 78 | |
| }, | |
| { | |
| "loss": 1.992941633943701e-06, | |
| "grad_norm": 9.314135240856558e-05, | |
| "learning_rate": 4.111111111111111e-06, | |
| "num_tokens": 379824.0, | |
| "completions/mean_length": 79.75, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 96.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.75, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 96.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.75, | |
| "kl": 0.0019929417176172137, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.2633333333333333, | |
| "step": 79 | |
| }, | |
| { | |
| "loss": 5.8060674490434394e-08, | |
| "grad_norm": 2.9884075047448277e-05, | |
| "learning_rate": 4.092592592592593e-06, | |
| "num_tokens": 382903.0, | |
| "completions/mean_length": 69.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 69.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 69.75, | |
| "kl": 5.8060667015524814e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.26666666666666666, | |
| "step": 80 | |
| }, | |
| { | |
| "loss": 2.2232532501220703e-05, | |
| "grad_norm": 0.9340193867683411, | |
| "learning_rate": 4.074074074074074e-06, | |
| "num_tokens": 389433.0, | |
| "completions/mean_length": 75.5, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.5, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 0.4749999940395355, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.4749999940395355, | |
| "reward_std": 0.34999996423721313, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 75.5, | |
| "kl": 0.022244371939450502, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.27, | |
| "step": 81 | |
| }, | |
| { | |
| "loss": 1.9052032484978554e-07, | |
| "grad_norm": 6.15498938714154e-05, | |
| "learning_rate": 4.055555555555556e-06, | |
| "num_tokens": 393170.0, | |
| "completions/mean_length": 82.25, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 86.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.25, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 86.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.25, | |
| "kl": 0.00019052031711908057, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.2733333333333333, | |
| "step": 82 | |
| }, | |
| { | |
| "loss": 3.3744271377145196e-07, | |
| "grad_norm": 9.197735198540613e-05, | |
| "learning_rate": 4.037037037037037e-06, | |
| "num_tokens": 396908.0, | |
| "completions/mean_length": 82.5, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.5, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.5, | |
| "kl": 0.0003374427105882205, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.27666666666666667, | |
| "step": 83 | |
| }, | |
| { | |
| "loss": 3.2648444175720215e-05, | |
| "grad_norm": 1.035704493522644, | |
| "learning_rate": 4.018518518518519e-06, | |
| "num_tokens": 403406.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 0.6499999761581421, | |
| "rewards/reward_fn/std": 0.40414518117904663, | |
| "reward": 0.6499999761581421, | |
| "reward_std": 0.40414518117904663, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 67.5, | |
| "kl": 0.032706082332879305, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.28, | |
| "step": 84 | |
| }, | |
| { | |
| "loss": 1.6296494322887156e-06, | |
| "grad_norm": 0.00016962463269010186, | |
| "learning_rate": 4.000000000000001e-06, | |
| "num_tokens": 406692.0, | |
| "completions/mean_length": 96.5, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 106.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.5, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 106.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.5, | |
| "kl": 0.001629649312235415, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.2833333333333333, | |
| "step": 85 | |
| }, | |
| { | |
| "loss": 7.547411229325007e-08, | |
| "grad_norm": 4.339437873568386e-05, | |
| "learning_rate": 3.9814814814814814e-06, | |
| "num_tokens": 409773.0, | |
| "completions/mean_length": 70.25, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 81.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.25, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 81.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.25, | |
| "kl": 7.547410586994374e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.2866666666666667, | |
| "step": 86 | |
| }, | |
| { | |
| "loss": 3.204380618626601e-07, | |
| "grad_norm": 9.214194142259657e-05, | |
| "learning_rate": 3.962962962962963e-06, | |
| "num_tokens": 413443.0, | |
| "completions/mean_length": 93.5, | |
| "completions/min_length": 81.0, | |
| "completions/max_length": 103.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.5, | |
| "completions/min_terminated_length": 81.0, | |
| "completions/max_terminated_length": 103.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.5, | |
| "kl": 0.0003204380627721548, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.29, | |
| "step": 87 | |
| }, | |
| { | |
| "loss": 3.7223100662231445e-05, | |
| "grad_norm": 0.8802705407142639, | |
| "learning_rate": 3.944444444444445e-06, | |
| "num_tokens": 419971.0, | |
| "completions/mean_length": 75.0, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.0, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 75.0, | |
| "kl": 0.03731195232830942, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.29333333333333333, | |
| "step": 88 | |
| }, | |
| { | |
| "loss": 2.1214867729213438e-07, | |
| "grad_norm": 6.772110646124929e-05, | |
| "learning_rate": 3.925925925925926e-06, | |
| "num_tokens": 423769.0, | |
| "completions/mean_length": 97.5, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 109.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 97.5, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 109.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 97.5, | |
| "kl": 0.00021214867228991352, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.2966666666666667, | |
| "step": 89 | |
| }, | |
| { | |
| "loss": 1.0971891128974676e-07, | |
| "grad_norm": 5.7245371863245964e-05, | |
| "learning_rate": 3.907407407407408e-06, | |
| "num_tokens": 426831.0, | |
| "completions/mean_length": 65.5, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.5, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.5, | |
| "kl": 0.0001097188978746999, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3, | |
| "step": 90 | |
| }, | |
| { | |
| "loss": 5.68784980714554e-06, | |
| "grad_norm": 0.00013987929560244083, | |
| "learning_rate": 3.88888888888889e-06, | |
| "num_tokens": 431033.0, | |
| "completions/mean_length": 89.5, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 105.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.5, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 105.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.5, | |
| "kl": 0.005687849479727447, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.30333333333333334, | |
| "step": 91 | |
| }, | |
| { | |
| "loss": 2.057557367152185e-06, | |
| "grad_norm": 0.00017590871721040457, | |
| "learning_rate": 3.87037037037037e-06, | |
| "num_tokens": 435175.0, | |
| "completions/mean_length": 64.5, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.5, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.5, | |
| "kl": 0.0020575573144014925, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.30666666666666664, | |
| "step": 92 | |
| }, | |
| { | |
| "loss": 7.651746273040771e-06, | |
| "grad_norm": 0.6655512452125549, | |
| "learning_rate": 3.851851851851852e-06, | |
| "num_tokens": 442749.0, | |
| "completions/mean_length": 91.5, | |
| "completions/min_length": 82.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 91.5, | |
| "completions/min_terminated_length": 82.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 91.5, | |
| "kl": 0.0076841484842589125, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.31, | |
| "step": 93 | |
| }, | |
| { | |
| "loss": 9.719022386889264e-08, | |
| "grad_norm": 5.316199894878082e-05, | |
| "learning_rate": 3.833333333333334e-06, | |
| "num_tokens": 445804.0, | |
| "completions/mean_length": 63.75, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 66.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 63.75, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 66.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 63.75, | |
| "kl": 9.719022182252957e-05, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.31333333333333335, | |
| "step": 94 | |
| }, | |
| { | |
| "loss": 2.8720914997393265e-06, | |
| "grad_norm": 0.00012377827079035342, | |
| "learning_rate": 3.814814814814815e-06, | |
| "num_tokens": 449888.0, | |
| "completions/mean_length": 66.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 67.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 67.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.0, | |
| "kl": 0.002872090961318463, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.31666666666666665, | |
| "step": 95 | |
| }, | |
| { | |
| "loss": 4.792213439941406e-05, | |
| "grad_norm": 0.7196746468544006, | |
| "learning_rate": 3.796296296296297e-06, | |
| "num_tokens": 456420.0, | |
| "completions/mean_length": 76.0, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 76.0, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 0.5, | |
| "rewards/reward_fn/std": 0.6271629333496094, | |
| "reward": 0.5, | |
| "reward_std": 0.6271628737449646, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 76.0, | |
| "kl": 0.0479273060336709, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.32, | |
| "step": 96 | |
| }, | |
| { | |
| "loss": 9.704103831609245e-07, | |
| "grad_norm": 0.00022661879484076053, | |
| "learning_rate": 3.777777777777778e-06, | |
| "num_tokens": 461779.0, | |
| "completions/mean_length": 74.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 94.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 74.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 94.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 74.75, | |
| "kl": 0.0009704103431431577, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3233333333333333, | |
| "step": 97 | |
| }, | |
| { | |
| "loss": 1.2592768143804278e-06, | |
| "grad_norm": 0.0003953844425268471, | |
| "learning_rate": 3.7592592592592597e-06, | |
| "num_tokens": 467103.0, | |
| "completions/mean_length": 66.0, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 66.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.0, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 66.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.0, | |
| "kl": 0.0012592768034664914, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.32666666666666666, | |
| "step": 98 | |
| }, | |
| { | |
| "loss": 6.228109441508423e-07, | |
| "grad_norm": 9.139542089542374e-05, | |
| "learning_rate": 3.740740740740741e-06, | |
| "num_tokens": 474696.0, | |
| "completions/mean_length": 96.25, | |
| "completions/min_length": 87.0, | |
| "completions/max_length": 111.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.25, | |
| "completions/min_terminated_length": 87.0, | |
| "completions/max_terminated_length": 111.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.25, | |
| "kl": 0.0006228109414223582, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.33, | |
| "step": 99 | |
| }, | |
| { | |
| "loss": 8.756982197155594e-07, | |
| "grad_norm": 0.00021289817232172936, | |
| "learning_rate": 3.7222222222222225e-06, | |
| "num_tokens": 478308.0, | |
| "completions/mean_length": 79.0, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 97.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.0, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 97.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.0, | |
| "kl": 0.0008756982861086726, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3333333333333333, | |
| "step": 100 | |
| }, | |
| { | |
| "loss": 6.360454426612705e-05, | |
| "grad_norm": 0.0009841566206887364, | |
| "learning_rate": 3.7037037037037037e-06, | |
| "num_tokens": 484819.0, | |
| "completions/mean_length": 70.75, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.75, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.75, | |
| "kl": 0.06360454112291336, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.33666666666666667, | |
| "step": 101 | |
| }, | |
| { | |
| "loss": 6.763004876120249e-07, | |
| "grad_norm": 0.00018609774997457862, | |
| "learning_rate": 3.6851851851851854e-06, | |
| "num_tokens": 488472.0, | |
| "completions/mean_length": 89.25, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.25, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.25, | |
| "kl": 0.0006763004057575017, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.34, | |
| "step": 102 | |
| }, | |
| { | |
| "loss": 1.9789573002526595e-07, | |
| "grad_norm": 5.712790516554378e-05, | |
| "learning_rate": 3.6666666666666666e-06, | |
| "num_tokens": 492230.0, | |
| "completions/mean_length": 87.5, | |
| "completions/min_length": 81.0, | |
| "completions/max_length": 92.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.5, | |
| "completions/min_terminated_length": 81.0, | |
| "completions/max_terminated_length": 92.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.5, | |
| "kl": 0.00019789573525486048, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3433333333333333, | |
| "step": 103 | |
| }, | |
| { | |
| "loss": 2.103467750202981e-06, | |
| "grad_norm": 0.00019168489961884916, | |
| "learning_rate": 3.6481481481481486e-06, | |
| "num_tokens": 495513.0, | |
| "completions/mean_length": 95.75, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 108.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 95.75, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 108.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 95.75, | |
| "kl": 0.002103467588312924, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3466666666666667, | |
| "step": 104 | |
| }, | |
| { | |
| "loss": 6.182389915920794e-05, | |
| "grad_norm": 0.0006671014707535505, | |
| "learning_rate": 3.6296296296296302e-06, | |
| "num_tokens": 502032.0, | |
| "completions/mean_length": 72.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 79.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 79.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.75, | |
| "kl": 0.06182389985769987, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.35, | |
| "step": 105 | |
| }, | |
| { | |
| "loss": 9.018430091600749e-07, | |
| "grad_norm": 0.00010332284000469372, | |
| "learning_rate": 3.6111111111111115e-06, | |
| "num_tokens": 509655.0, | |
| "completions/mean_length": 103.75, | |
| "completions/min_length": 82.0, | |
| "completions/max_length": 114.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 103.75, | |
| "completions/min_terminated_length": 82.0, | |
| "completions/max_terminated_length": 114.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 103.75, | |
| "kl": 0.0009018429482239299, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.35333333333333333, | |
| "step": 106 | |
| }, | |
| { | |
| "loss": 6.047176066203974e-05, | |
| "grad_norm": 0.0004248884506523609, | |
| "learning_rate": 3.592592592592593e-06, | |
| "num_tokens": 516178.0, | |
| "completions/mean_length": 73.75, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 81.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.75, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 81.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.75, | |
| "kl": 0.06047175545245409, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3566666666666667, | |
| "step": 107 | |
| }, | |
| { | |
| "loss": 3.385654565590812e-07, | |
| "grad_norm": 0.00015579280443489552, | |
| "learning_rate": 3.5740740740740743e-06, | |
| "num_tokens": 519262.0, | |
| "completions/mean_length": 71.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 90.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 90.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.0, | |
| "kl": 0.00033856545906019164, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.36, | |
| "step": 108 | |
| }, | |
| { | |
| "loss": 1.3627701491714106e-06, | |
| "grad_norm": 0.00012413490912877023, | |
| "learning_rate": 3.555555555555556e-06, | |
| "num_tokens": 522570.0, | |
| "completions/mean_length": 102.0, | |
| "completions/min_length": 92.0, | |
| "completions/max_length": 113.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 102.0, | |
| "completions/min_terminated_length": 92.0, | |
| "completions/max_terminated_length": 113.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 102.0, | |
| "kl": 0.0013627700682263821, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.36333333333333334, | |
| "step": 109 | |
| }, | |
| { | |
| "loss": 6.167578976601362e-05, | |
| "grad_norm": 0.00043891475070267916, | |
| "learning_rate": 3.537037037037037e-06, | |
| "num_tokens": 529091.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 0.0616757906973362, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.36666666666666664, | |
| "step": 110 | |
| }, | |
| { | |
| "loss": 2.1278858184814453e-05, | |
| "grad_norm": 2.589689016342163, | |
| "learning_rate": 3.5185185185185187e-06, | |
| "num_tokens": 533200.0, | |
| "completions/mean_length": 67.25, | |
| "completions/min_length": 56.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.25, | |
| "completions/min_terminated_length": 56.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 67.25, | |
| "kl": 0.021340800682082772, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.37, | |
| "step": 111 | |
| }, | |
| { | |
| "loss": 1.1418839562793437e-07, | |
| "grad_norm": 5.785573739558458e-05, | |
| "learning_rate": 3.5e-06, | |
| "num_tokens": 536260.0, | |
| "completions/mean_length": 65.0, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 71.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.0, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 71.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.0, | |
| "kl": 0.00011418838585086633, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.37333333333333335, | |
| "step": 112 | |
| }, | |
| { | |
| "loss": 3.573298954506754e-07, | |
| "grad_norm": 0.00010180289973504841, | |
| "learning_rate": 3.481481481481482e-06, | |
| "num_tokens": 540021.0, | |
| "completions/mean_length": 88.25, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 112.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 88.25, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 112.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 88.25, | |
| "kl": 0.0003573299036361277, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.37666666666666665, | |
| "step": 113 | |
| }, | |
| { | |
| "loss": 1.8434477624396095e-06, | |
| "grad_norm": 0.00022729912598151714, | |
| "learning_rate": 3.4629629629629628e-06, | |
| "num_tokens": 547564.0, | |
| "completions/mean_length": 83.75, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 86.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.75, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 86.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.75, | |
| "kl": 0.0018434475641697645, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.38, | |
| "step": 114 | |
| }, | |
| { | |
| "loss": 2.0494962882366963e-06, | |
| "grad_norm": 0.0002225613861810416, | |
| "learning_rate": 3.444444444444445e-06, | |
| "num_tokens": 551734.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.5, | |
| "kl": 0.0020494962809607387, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.38333333333333336, | |
| "step": 115 | |
| }, | |
| { | |
| "loss": 1.4116823194854078e-06, | |
| "grad_norm": 0.0001264088787138462, | |
| "learning_rate": 3.4259259259259265e-06, | |
| "num_tokens": 555019.0, | |
| "completions/mean_length": 96.25, | |
| "completions/min_length": 88.0, | |
| "completions/max_length": 103.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.25, | |
| "completions/min_terminated_length": 88.0, | |
| "completions/max_terminated_length": 103.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.25, | |
| "kl": 0.0014116823440417647, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.38666666666666666, | |
| "step": 116 | |
| }, | |
| { | |
| "loss": 8.412856686845771e-07, | |
| "grad_norm": 9.905538900056854e-05, | |
| "learning_rate": 3.4074074074074077e-06, | |
| "num_tokens": 562629.0, | |
| "completions/mean_length": 100.5, | |
| "completions/min_length": 90.0, | |
| "completions/max_length": 113.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 100.5, | |
| "completions/min_terminated_length": 90.0, | |
| "completions/max_terminated_length": 113.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 100.5, | |
| "kl": 0.0008412855677306652, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.39, | |
| "step": 117 | |
| }, | |
| { | |
| "loss": 1.4454126358032227e-05, | |
| "grad_norm": 2.6465179920196533, | |
| "learning_rate": 3.3888888888888893e-06, | |
| "num_tokens": 566776.0, | |
| "completions/mean_length": 78.75, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 78.75, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 78.75, | |
| "kl": 0.0145443812943995, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.3933333333333333, | |
| "step": 118 | |
| }, | |
| { | |
| "loss": 6.922218744875863e-07, | |
| "grad_norm": 8.306170639116317e-05, | |
| "learning_rate": 3.3703703703703705e-06, | |
| "num_tokens": 574362.0, | |
| "completions/mean_length": 94.5, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 107.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 94.5, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 107.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 94.5, | |
| "kl": 0.0006922218672116287, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.39666666666666667, | |
| "step": 119 | |
| }, | |
| { | |
| "loss": 7.00816372045665e-07, | |
| "grad_norm": 0.00016947407857514918, | |
| "learning_rate": 3.351851851851852e-06, | |
| "num_tokens": 578043.0, | |
| "completions/mean_length": 96.25, | |
| "completions/min_length": 93.0, | |
| "completions/max_length": 101.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.25, | |
| "completions/min_terminated_length": 93.0, | |
| "completions/max_terminated_length": 101.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.25, | |
| "kl": 0.000700816344760824, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4, | |
| "step": 120 | |
| }, | |
| { | |
| "loss": 6.2500530475517735e-06, | |
| "grad_norm": 0.0015128880040720105, | |
| "learning_rate": 3.3333333333333333e-06, | |
| "num_tokens": 583936.0, | |
| "completions/mean_length": 88.25, | |
| "completions/min_length": 81.0, | |
| "completions/max_length": 96.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 88.25, | |
| "completions/min_terminated_length": 81.0, | |
| "completions/max_terminated_length": 96.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 88.25, | |
| "kl": 0.006250052712857723, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4033333333333333, | |
| "step": 121 | |
| }, | |
| { | |
| "loss": 1.2354882983345306e-07, | |
| "grad_norm": 4.0171587897930294e-05, | |
| "learning_rate": 3.314814814814815e-06, | |
| "num_tokens": 587704.0, | |
| "completions/mean_length": 90.0, | |
| "completions/min_length": 87.0, | |
| "completions/max_length": 96.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.0, | |
| "completions/min_terminated_length": 87.0, | |
| "completions/max_terminated_length": 96.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 90.0, | |
| "kl": 0.00012354881801002193, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4066666666666667, | |
| "step": 122 | |
| }, | |
| { | |
| "loss": 5.609046638710424e-06, | |
| "grad_norm": 0.0014228485524654388, | |
| "learning_rate": 3.296296296296296e-06, | |
| "num_tokens": 593064.0, | |
| "completions/mean_length": 75.0, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 102.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.0, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 102.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 75.0, | |
| "kl": 0.005609046726021916, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.41, | |
| "step": 123 | |
| }, | |
| { | |
| "loss": 5.86629867029842e-05, | |
| "grad_norm": 0.0002953781222458929, | |
| "learning_rate": 3.277777777777778e-06, | |
| "num_tokens": 599591.0, | |
| "completions/mean_length": 74.75, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 74.75, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 74.75, | |
| "kl": 0.058662984520196915, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.41333333333333333, | |
| "step": 124 | |
| }, | |
| { | |
| "loss": 6.252167077036574e-05, | |
| "grad_norm": 0.0003890777297783643, | |
| "learning_rate": 3.25925925925926e-06, | |
| "num_tokens": 606105.0, | |
| "completions/mean_length": 71.5, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.5, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.5, | |
| "kl": 0.06252166721969843, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4166666666666667, | |
| "step": 125 | |
| }, | |
| { | |
| "loss": 8.384186912735458e-06, | |
| "grad_norm": 0.00043141015339642763, | |
| "learning_rate": 3.240740740740741e-06, | |
| "num_tokens": 610273.0, | |
| "completions/mean_length": 75.0, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.0, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 75.0, | |
| "kl": 0.008384186425246298, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.42, | |
| "step": 126 | |
| }, | |
| { | |
| "loss": 4.36719801655272e-06, | |
| "grad_norm": 0.0008875001221895218, | |
| "learning_rate": 3.2222222222222227e-06, | |
| "num_tokens": 615628.0, | |
| "completions/mean_length": 73.75, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 95.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.75, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 95.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.75, | |
| "kl": 0.004367197921965271, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.42333333333333334, | |
| "step": 127 | |
| }, | |
| { | |
| "loss": 1.1651266049739206e-06, | |
| "grad_norm": 0.00011579222336877137, | |
| "learning_rate": 3.203703703703704e-06, | |
| "num_tokens": 623224.0, | |
| "completions/mean_length": 97.0, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 113.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 97.0, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 113.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 97.0, | |
| "kl": 0.0011651265158434398, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4266666666666667, | |
| "step": 128 | |
| }, | |
| { | |
| "loss": 2.0292654880904593e-06, | |
| "grad_norm": 0.00023187890474218875, | |
| "learning_rate": 3.1851851851851855e-06, | |
| "num_tokens": 630775.0, | |
| "completions/mean_length": 85.75, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 102.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 85.75, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 102.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 85.75, | |
| "kl": 0.002029265306191519, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.43, | |
| "step": 129 | |
| }, | |
| { | |
| "loss": 5.629781867355632e-07, | |
| "grad_norm": 0.0001309424260398373, | |
| "learning_rate": 3.1666666666666667e-06, | |
| "num_tokens": 634432.0, | |
| "completions/mean_length": 90.25, | |
| "completions/min_length": 81.0, | |
| "completions/max_length": 98.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.25, | |
| "completions/min_terminated_length": 81.0, | |
| "completions/max_terminated_length": 98.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 90.25, | |
| "kl": 0.0005629781699099112, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.43333333333333335, | |
| "step": 130 | |
| }, | |
| { | |
| "loss": 9.380090659760754e-07, | |
| "grad_norm": 0.00010971970914397389, | |
| "learning_rate": 3.1481481481481483e-06, | |
| "num_tokens": 642100.0, | |
| "completions/mean_length": 115.0, | |
| "completions/min_length": 92.0, | |
| "completions/max_length": 143.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 115.0, | |
| "completions/min_terminated_length": 92.0, | |
| "completions/max_terminated_length": 143.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 115.0, | |
| "kl": 0.000938009048695676, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.43666666666666665, | |
| "step": 131 | |
| }, | |
| { | |
| "loss": 6.613444384129252e-06, | |
| "grad_norm": 0.0015011809300631285, | |
| "learning_rate": 3.1296296296296295e-06, | |
| "num_tokens": 647453.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 86.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 86.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 0.006613444013055414, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.44, | |
| "step": 132 | |
| }, | |
| { | |
| "loss": 1.1578202247619629e-05, | |
| "grad_norm": 1.2958519458770752, | |
| "learning_rate": 3.1111111111111116e-06, | |
| "num_tokens": 653332.0, | |
| "completions/mean_length": 84.75, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 84.75, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 84.75, | |
| "kl": 0.011619423516094685, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.44333333333333336, | |
| "step": 133 | |
| }, | |
| { | |
| "loss": 3.739810381375719e-07, | |
| "grad_norm": 8.853741019265726e-05, | |
| "learning_rate": 3.0925925925925928e-06, | |
| "num_tokens": 656927.0, | |
| "completions/mean_length": 74.75, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 74.75, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 74.75, | |
| "kl": 0.0003739810417755507, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.44666666666666666, | |
| "step": 134 | |
| }, | |
| { | |
| "loss": 2.107486034219619e-05, | |
| "grad_norm": 0.004327178932726383, | |
| "learning_rate": 3.0740740740740744e-06, | |
| "num_tokens": 662783.0, | |
| "completions/mean_length": 79.0, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.0, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.0, | |
| "kl": 0.021074859891086817, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.45, | |
| "step": 135 | |
| }, | |
| { | |
| "loss": 1.576767544975155e-06, | |
| "grad_norm": 0.00017685597413219512, | |
| "learning_rate": 3.055555555555556e-06, | |
| "num_tokens": 670360.0, | |
| "completions/mean_length": 92.25, | |
| "completions/min_length": 83.0, | |
| "completions/max_length": 98.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 92.25, | |
| "completions/min_terminated_length": 83.0, | |
| "completions/max_terminated_length": 98.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 92.25, | |
| "kl": 0.0015767674485687166, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4533333333333333, | |
| "step": 136 | |
| }, | |
| { | |
| "loss": 2.1992673282511532e-06, | |
| "grad_norm": 0.00021581645705737174, | |
| "learning_rate": 3.0370370370370372e-06, | |
| "num_tokens": 677949.0, | |
| "completions/mean_length": 95.25, | |
| "completions/min_length": 82.0, | |
| "completions/max_length": 109.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 95.25, | |
| "completions/min_terminated_length": 82.0, | |
| "completions/max_terminated_length": 109.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 95.25, | |
| "kl": 0.002199267524702009, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.45666666666666667, | |
| "step": 137 | |
| }, | |
| { | |
| "loss": 6.151283287181286e-07, | |
| "grad_norm": 9.6255520475097e-05, | |
| "learning_rate": 3.018518518518519e-06, | |
| "num_tokens": 685586.0, | |
| "completions/mean_length": 107.25, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 122.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 107.25, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 122.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 107.25, | |
| "kl": 0.0006151282832433935, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.46, | |
| "step": 138 | |
| }, | |
| { | |
| "loss": 3.0734272513655014e-06, | |
| "grad_norm": 0.00104941101744771, | |
| "learning_rate": 3e-06, | |
| "num_tokens": 688847.0, | |
| "completions/mean_length": 90.25, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 108.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.25, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 108.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 90.25, | |
| "kl": 0.0030734269530512393, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4633333333333333, | |
| "step": 139 | |
| }, | |
| { | |
| "loss": 2.1971184196445392e-06, | |
| "grad_norm": 0.00010507876868359745, | |
| "learning_rate": 2.9814814814814817e-06, | |
| "num_tokens": 693018.0, | |
| "completions/mean_length": 71.75, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 81.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.75, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 81.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.75, | |
| "kl": 0.002197118563344702, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4666666666666667, | |
| "step": 140 | |
| }, | |
| { | |
| "loss": 5.22006985193002e-06, | |
| "grad_norm": 0.0009780385298654437, | |
| "learning_rate": 2.962962962962963e-06, | |
| "num_tokens": 696260.0, | |
| "completions/mean_length": 85.5, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 85.5, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 85.5, | |
| "kl": 0.005220069288043305, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.47, | |
| "step": 141 | |
| }, | |
| { | |
| "loss": 1.874823510661372e-06, | |
| "grad_norm": 0.00015883729793131351, | |
| "learning_rate": 2.944444444444445e-06, | |
| "num_tokens": 699571.0, | |
| "completions/mean_length": 102.75, | |
| "completions/min_length": 97.0, | |
| "completions/max_length": 115.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 102.75, | |
| "completions/min_terminated_length": 97.0, | |
| "completions/max_terminated_length": 115.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 102.75, | |
| "kl": 0.0018748235597740859, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.47333333333333333, | |
| "step": 142 | |
| }, | |
| { | |
| "loss": 6.359740382322343e-06, | |
| "grad_norm": 0.0016190956812351942, | |
| "learning_rate": 2.9259259259259257e-06, | |
| "num_tokens": 704903.0, | |
| "completions/mean_length": 68.0, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.0, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 68.0, | |
| "kl": 0.006359739985782653, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4766666666666667, | |
| "step": 143 | |
| }, | |
| { | |
| "loss": 2.48378796641191e-07, | |
| "grad_norm": 8.51371805765666e-05, | |
| "learning_rate": 2.907407407407408e-06, | |
| "num_tokens": 708647.0, | |
| "completions/mean_length": 84.0, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 90.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 84.0, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 90.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 84.0, | |
| "kl": 0.0002483787930032122, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.48, | |
| "step": 144 | |
| }, | |
| { | |
| "loss": 1.328811686107656e-05, | |
| "grad_norm": 0.0001837281888583675, | |
| "learning_rate": 2.888888888888889e-06, | |
| "num_tokens": 714517.0, | |
| "completions/mean_length": 82.5, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 93.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.5, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 93.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.5, | |
| "kl": 0.013288116315379739, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.48333333333333334, | |
| "step": 145 | |
| }, | |
| { | |
| "loss": 9.009476116261794e-07, | |
| "grad_norm": 0.00011331303539918736, | |
| "learning_rate": 2.8703703703703706e-06, | |
| "num_tokens": 722085.0, | |
| "completions/mean_length": 90.0, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 96.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.0, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 96.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 90.0, | |
| "kl": 0.0009009475616039708, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.4866666666666667, | |
| "step": 146 | |
| }, | |
| { | |
| "loss": 6.697333901684033e-06, | |
| "grad_norm": 0.00017503471462987363, | |
| "learning_rate": 2.8518518518518522e-06, | |
| "num_tokens": 726211.0, | |
| "completions/mean_length": 66.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 69.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 69.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.5, | |
| "kl": 0.0066973339999094605, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.49, | |
| "step": 147 | |
| }, | |
| { | |
| "loss": 2.860901986423414e-07, | |
| "grad_norm": 0.0001681904832366854, | |
| "learning_rate": 2.8333333333333335e-06, | |
| "num_tokens": 729280.0, | |
| "completions/mean_length": 67.25, | |
| "completions/min_length": 59.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.25, | |
| "completions/min_terminated_length": 59.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.25, | |
| "kl": 0.0002860901895473944, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.49333333333333335, | |
| "step": 148 | |
| }, | |
| { | |
| "loss": 7.192904740804806e-05, | |
| "grad_norm": 0.002006913535296917, | |
| "learning_rate": 2.814814814814815e-06, | |
| "num_tokens": 735799.0, | |
| "completions/mean_length": 72.75, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.75, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.75, | |
| "kl": 0.07192904874682426, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.49666666666666665, | |
| "step": 149 | |
| }, | |
| { | |
| "loss": 5.770430652773939e-05, | |
| "grad_norm": 0.0021646975073963404, | |
| "learning_rate": 2.7962962962962963e-06, | |
| "num_tokens": 742348.0, | |
| "completions/mean_length": 80.25, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 90.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.25, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 90.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.25, | |
| "kl": 0.057704306207597256, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5, | |
| "step": 150 | |
| }, | |
| { | |
| "loss": 1.312791027885396e-05, | |
| "grad_norm": 0.0001467515539843589, | |
| "learning_rate": 2.7777777777777783e-06, | |
| "num_tokens": 748217.0, | |
| "completions/mean_length": 82.25, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.25, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.25, | |
| "kl": 0.013127909740433097, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5033333333333333, | |
| "step": 151 | |
| }, | |
| { | |
| "loss": 5.624011464533396e-05, | |
| "grad_norm": 0.00024933615350164473, | |
| "learning_rate": 2.759259259259259e-06, | |
| "num_tokens": 754761.0, | |
| "completions/mean_length": 79.0, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.0, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.0, | |
| "kl": 0.05624010972678661, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5066666666666667, | |
| "step": 152 | |
| }, | |
| { | |
| "loss": 4.969537258148193e-06, | |
| "grad_norm": 1.194063663482666, | |
| "learning_rate": 2.740740740740741e-06, | |
| "num_tokens": 758055.0, | |
| "completions/mean_length": 98.5, | |
| "completions/min_length": 84.0, | |
| "completions/max_length": 111.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 98.5, | |
| "completions/min_terminated_length": 84.0, | |
| "completions/max_terminated_length": 111.0, | |
| "rewards/reward_fn/mean": 0.75, | |
| "rewards/reward_fn/std": 2.5, | |
| "reward": 0.75, | |
| "reward_std": 2.5, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 98.5, | |
| "kl": 0.004997963987989351, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.51, | |
| "step": 153 | |
| }, | |
| { | |
| "loss": 6.250140722841024e-05, | |
| "grad_norm": 0.0016695293597877026, | |
| "learning_rate": 2.7222222222222224e-06, | |
| "num_tokens": 764614.0, | |
| "completions/mean_length": 82.75, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 104.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.75, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 104.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.75, | |
| "kl": 0.06250140257179737, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5133333333333333, | |
| "step": 154 | |
| }, | |
| { | |
| "loss": 8.424445695709437e-06, | |
| "grad_norm": 0.0021802084520459175, | |
| "learning_rate": 2.703703703703704e-06, | |
| "num_tokens": 769948.0, | |
| "completions/mean_length": 68.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 68.5, | |
| "kl": 0.008424444939009845, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5166666666666667, | |
| "step": 155 | |
| }, | |
| { | |
| "loss": 1.388172904626117e-06, | |
| "grad_norm": 0.00012273604806978256, | |
| "learning_rate": 2.6851851851851856e-06, | |
| "num_tokens": 774112.0, | |
| "completions/mean_length": 69.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 76.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 69.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 76.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 69.0, | |
| "kl": 0.0013881728518754244, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.52, | |
| "step": 156 | |
| }, | |
| { | |
| "loss": 1.9816582152998308e-07, | |
| "grad_norm": 0.0001281750010093674, | |
| "learning_rate": 2.666666666666667e-06, | |
| "num_tokens": 777170.0, | |
| "completions/mean_length": 64.5, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.5, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.5, | |
| "kl": 0.00019816579879261553, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5233333333333333, | |
| "step": 157 | |
| }, | |
| { | |
| "loss": 1.906811485241633e-05, | |
| "grad_norm": 0.00014608577475883067, | |
| "learning_rate": 2.6481481481481485e-06, | |
| "num_tokens": 781297.0, | |
| "completions/mean_length": 71.75, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.75, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.75, | |
| "kl": 0.019068114459514618, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5266666666666666, | |
| "step": 158 | |
| }, | |
| { | |
| "loss": 1.2931917581227026e-06, | |
| "grad_norm": 0.0003522019542288035, | |
| "learning_rate": 2.6296296296296297e-06, | |
| "num_tokens": 784916.0, | |
| "completions/mean_length": 80.75, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.75, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.75, | |
| "kl": 0.0012931917735841125, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.53, | |
| "step": 159 | |
| }, | |
| { | |
| "loss": 3.2540967822569655e-06, | |
| "grad_norm": 0.00016952218720689416, | |
| "learning_rate": 2.6111111111111113e-06, | |
| "num_tokens": 790257.0, | |
| "completions/mean_length": 70.25, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.25, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.25, | |
| "kl": 0.0032540965476073325, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5333333333333333, | |
| "step": 160 | |
| }, | |
| { | |
| "loss": 8.211204090002866e-07, | |
| "grad_norm": 0.00012330037134233862, | |
| "learning_rate": 2.5925925925925925e-06, | |
| "num_tokens": 797871.0, | |
| "completions/mean_length": 101.5, | |
| "completions/min_length": 92.0, | |
| "completions/max_length": 106.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 101.5, | |
| "completions/min_terminated_length": 92.0, | |
| "completions/max_terminated_length": 106.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 101.5, | |
| "kl": 0.0008211203166865744, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5366666666666666, | |
| "step": 161 | |
| }, | |
| { | |
| "loss": 1.375394276692532e-05, | |
| "grad_norm": 0.00013194767234381288, | |
| "learning_rate": 2.5740740740740745e-06, | |
| "num_tokens": 803718.0, | |
| "completions/mean_length": 76.75, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 76.75, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 76.75, | |
| "kl": 0.01375394081696868, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.54, | |
| "step": 162 | |
| }, | |
| { | |
| "loss": 1.0366624110247358e-06, | |
| "grad_norm": 0.0001573658228153363, | |
| "learning_rate": 2.5555555555555557e-06, | |
| "num_tokens": 811304.0, | |
| "completions/mean_length": 94.5, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 113.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 94.5, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 113.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 94.5, | |
| "kl": 0.0010366624264861457, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5433333333333333, | |
| "step": 163 | |
| }, | |
| { | |
| "loss": 1.287557006435236e-05, | |
| "grad_norm": 9.028160275192931e-05, | |
| "learning_rate": 2.5370370370370374e-06, | |
| "num_tokens": 817169.0, | |
| "completions/mean_length": 81.25, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 81.25, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 81.25, | |
| "kl": 0.012875569052994251, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5466666666666666, | |
| "step": 164 | |
| }, | |
| { | |
| "loss": 2.0683904722318402e-07, | |
| "grad_norm": 5.789366696262732e-05, | |
| "learning_rate": 2.5185185185185186e-06, | |
| "num_tokens": 820913.0, | |
| "completions/mean_length": 84.0, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 89.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 84.0, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 89.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 84.0, | |
| "kl": 0.00020683901857410092, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.55, | |
| "step": 165 | |
| }, | |
| { | |
| "loss": 1.2532956134236883e-05, | |
| "grad_norm": 0.00018154211284127086, | |
| "learning_rate": 2.5e-06, | |
| "num_tokens": 825018.0, | |
| "completions/mean_length": 65.25, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.25, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.25, | |
| "kl": 0.012532955268397927, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5533333333333333, | |
| "step": 166 | |
| }, | |
| { | |
| "loss": 6.076489080442116e-05, | |
| "grad_norm": 0.00032158708199858665, | |
| "learning_rate": 2.481481481481482e-06, | |
| "num_tokens": 831533.0, | |
| "completions/mean_length": 71.75, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.75, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.75, | |
| "kl": 0.060764892026782036, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5566666666666666, | |
| "step": 167 | |
| }, | |
| { | |
| "loss": 2.6562815946817864e-07, | |
| "grad_norm": 6.129117537057027e-05, | |
| "learning_rate": 2.462962962962963e-06, | |
| "num_tokens": 835320.0, | |
| "completions/mean_length": 94.75, | |
| "completions/min_length": 82.0, | |
| "completions/max_length": 109.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 94.75, | |
| "completions/min_terminated_length": 82.0, | |
| "completions/max_terminated_length": 109.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 94.75, | |
| "kl": 0.0002656281503732316, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.56, | |
| "step": 168 | |
| }, | |
| { | |
| "loss": 1.4901161193847656e-07, | |
| "grad_norm": 0.6185609102249146, | |
| "learning_rate": 2.4444444444444447e-06, | |
| "num_tokens": 839091.0, | |
| "completions/mean_length": 90.75, | |
| "completions/min_length": 82.0, | |
| "completions/max_length": 106.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.75, | |
| "completions/min_terminated_length": 82.0, | |
| "completions/max_terminated_length": 106.0, | |
| "rewards/reward_fn/mean": 0.5, | |
| "rewards/reward_fn/std": 3.0, | |
| "reward": 0.5, | |
| "reward_std": 3.0, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 90.75, | |
| "kl": 0.0001545305331092095, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5633333333333334, | |
| "step": 169 | |
| }, | |
| { | |
| "loss": 5.510064511327073e-05, | |
| "grad_norm": 0.0002984635648317635, | |
| "learning_rate": 2.425925925925926e-06, | |
| "num_tokens": 845641.0, | |
| "completions/mean_length": 80.5, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 95.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.5, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 95.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.5, | |
| "kl": 0.055100646801292896, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5666666666666667, | |
| "step": 170 | |
| }, | |
| { | |
| "loss": 6.049393778084777e-05, | |
| "grad_norm": 0.0004177230584900826, | |
| "learning_rate": 2.4074074074074075e-06, | |
| "num_tokens": 852164.0, | |
| "completions/mean_length": 73.75, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.75, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.75, | |
| "kl": 0.060493932105600834, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.57, | |
| "step": 171 | |
| }, | |
| { | |
| "loss": 2.8759241104125977e-05, | |
| "grad_norm": 1.8797731399536133, | |
| "learning_rate": 2.388888888888889e-06, | |
| "num_tokens": 856247.0, | |
| "completions/mean_length": 67.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 71.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 71.0, | |
| "rewards/reward_fn/mean": 0.6499999761581421, | |
| "rewards/reward_fn/std": 0.40414518117904663, | |
| "reward": 0.6499999761581421, | |
| "reward_std": 0.40414518117904663, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 67.75, | |
| "kl": 0.028830823255702853, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5733333333333334, | |
| "step": 172 | |
| }, | |
| { | |
| "loss": 2.5884341994242277e-06, | |
| "grad_norm": 0.0001693519443506375, | |
| "learning_rate": 2.3703703703703707e-06, | |
| "num_tokens": 859518.0, | |
| "completions/mean_length": 92.75, | |
| "completions/min_length": 86.0, | |
| "completions/max_length": 97.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 92.75, | |
| "completions/min_terminated_length": 86.0, | |
| "completions/max_terminated_length": 97.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 92.75, | |
| "kl": 0.002588434348581359, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5766666666666667, | |
| "step": 173 | |
| }, | |
| { | |
| "loss": 4.576211722451262e-06, | |
| "grad_norm": 0.00015597045421600342, | |
| "learning_rate": 2.351851851851852e-06, | |
| "num_tokens": 863660.0, | |
| "completions/mean_length": 66.5, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.5, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.5, | |
| "kl": 0.004576211213134229, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.58, | |
| "step": 174 | |
| }, | |
| { | |
| "loss": 6.0711579862982035e-05, | |
| "grad_norm": 0.00035171982017345726, | |
| "learning_rate": 2.3333333333333336e-06, | |
| "num_tokens": 870174.0, | |
| "completions/mean_length": 71.5, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.5, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.5, | |
| "kl": 0.060711572878062725, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5833333333333334, | |
| "step": 175 | |
| }, | |
| { | |
| "loss": 6.116629811003804e-05, | |
| "grad_norm": 0.00019448304374236614, | |
| "learning_rate": 2.314814814814815e-06, | |
| "num_tokens": 876683.0, | |
| "completions/mean_length": 70.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 79.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 79.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.25, | |
| "kl": 0.061166295781731606, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5866666666666667, | |
| "step": 176 | |
| }, | |
| { | |
| "loss": 3.378382643859368e-06, | |
| "grad_norm": 0.00017247068171855062, | |
| "learning_rate": 2.2962962962962964e-06, | |
| "num_tokens": 882012.0, | |
| "completions/mean_length": 67.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.25, | |
| "kl": 0.0033783825929276645, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.59, | |
| "step": 177 | |
| }, | |
| { | |
| "loss": 1.1370960351086978e-07, | |
| "grad_norm": 6.698760989820585e-05, | |
| "learning_rate": 2.277777777777778e-06, | |
| "num_tokens": 885092.0, | |
| "completions/mean_length": 70.0, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.0, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.0, | |
| "kl": 0.00011370960419299081, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5933333333333334, | |
| "step": 178 | |
| }, | |
| { | |
| "loss": 2.751249667198863e-06, | |
| "grad_norm": 0.00024290291185025126, | |
| "learning_rate": 2.2592592592592592e-06, | |
| "num_tokens": 892650.0, | |
| "completions/mean_length": 87.5, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.5, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.5, | |
| "kl": 0.002751249528955668, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.5966666666666667, | |
| "step": 179 | |
| }, | |
| { | |
| "loss": 3.999042291980004e-06, | |
| "grad_norm": 0.0009320350945927203, | |
| "learning_rate": 2.240740740740741e-06, | |
| "num_tokens": 900265.0, | |
| "completions/mean_length": 101.75, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 117.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 101.75, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 117.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 101.75, | |
| "kl": 0.00399904276127927, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6, | |
| "step": 180 | |
| }, | |
| { | |
| "loss": 5.955396409262903e-05, | |
| "grad_norm": 0.0003322784323245287, | |
| "learning_rate": 2.222222222222222e-06, | |
| "num_tokens": 906793.0, | |
| "completions/mean_length": 75.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 75.0, | |
| "kl": 0.05955395940691233, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6033333333333334, | |
| "step": 181 | |
| }, | |
| { | |
| "loss": 1.3914278497395571e-05, | |
| "grad_norm": 0.00013249229232314974, | |
| "learning_rate": 2.203703703703704e-06, | |
| "num_tokens": 912641.0, | |
| "completions/mean_length": 77.0, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 77.0, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 77.0, | |
| "kl": 0.01391427731141448, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6066666666666667, | |
| "step": 182 | |
| }, | |
| { | |
| "loss": 2.101810423482675e-06, | |
| "grad_norm": 0.00011538797843968496, | |
| "learning_rate": 2.1851851851851853e-06, | |
| "num_tokens": 915930.0, | |
| "completions/mean_length": 97.25, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 112.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 97.25, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 112.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 97.25, | |
| "kl": 0.002101810270687565, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.61, | |
| "step": 183 | |
| }, | |
| { | |
| "loss": 4.058705599163659e-06, | |
| "grad_norm": 8.868318400345743e-05, | |
| "learning_rate": 2.166666666666667e-06, | |
| "num_tokens": 920059.0, | |
| "completions/mean_length": 64.25, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 66.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.25, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 66.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.25, | |
| "kl": 0.004058705526404083, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6133333333333333, | |
| "step": 184 | |
| }, | |
| { | |
| "loss": 1.080816900866921e-06, | |
| "grad_norm": 0.0001380756584694609, | |
| "learning_rate": 2.148148148148148e-06, | |
| "num_tokens": 927708.0, | |
| "completions/mean_length": 110.25, | |
| "completions/min_length": 105.0, | |
| "completions/max_length": 113.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 110.25, | |
| "completions/min_terminated_length": 105.0, | |
| "completions/max_terminated_length": 113.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 110.25, | |
| "kl": 0.00108081687358208, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6166666666666667, | |
| "step": 185 | |
| }, | |
| { | |
| "loss": 4.928518592350883e-06, | |
| "grad_norm": 0.00027522293385118246, | |
| "learning_rate": 2.1296296296296298e-06, | |
| "num_tokens": 931844.0, | |
| "completions/mean_length": 64.0, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.0, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.0, | |
| "kl": 0.00492851814487949, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.62, | |
| "step": 186 | |
| }, | |
| { | |
| "loss": 1.2626842362806201e-05, | |
| "grad_norm": 0.003474925644695759, | |
| "learning_rate": 2.1111111111111114e-06, | |
| "num_tokens": 937164.0, | |
| "completions/mean_length": 65.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 66.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 66.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.0, | |
| "kl": 0.01262684230459854, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6233333333333333, | |
| "step": 187 | |
| }, | |
| { | |
| "loss": 3.454735633567907e-06, | |
| "grad_norm": 0.00013449504331219941, | |
| "learning_rate": 2.0925925925925926e-06, | |
| "num_tokens": 941310.0, | |
| "completions/mean_length": 66.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.5, | |
| "kl": 0.0034547357354313135, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6266666666666667, | |
| "step": 188 | |
| }, | |
| { | |
| "loss": 6.316366489045322e-05, | |
| "grad_norm": 0.0005944407894276083, | |
| "learning_rate": 2.0740740740740742e-06, | |
| "num_tokens": 947823.0, | |
| "completions/mean_length": 71.25, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.25, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.25, | |
| "kl": 0.06316366419196129, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.63, | |
| "step": 189 | |
| }, | |
| { | |
| "loss": 9.276707714889199e-06, | |
| "grad_norm": 0.0018583047203719616, | |
| "learning_rate": 2.0555555555555555e-06, | |
| "num_tokens": 951082.0, | |
| "completions/mean_length": 89.75, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 102.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.75, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 102.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.75, | |
| "kl": 0.009276707976823673, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6333333333333333, | |
| "step": 190 | |
| }, | |
| { | |
| "loss": 1.2700134902843274e-05, | |
| "grad_norm": 0.00011148088378831744, | |
| "learning_rate": 2.037037037037037e-06, | |
| "num_tokens": 956959.0, | |
| "completions/mean_length": 84.25, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 84.25, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 84.25, | |
| "kl": 0.012700134422630072, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6366666666666667, | |
| "step": 191 | |
| }, | |
| { | |
| "loss": 2.4768547746134573e-07, | |
| "grad_norm": 0.00010171010217163712, | |
| "learning_rate": 2.0185185185185187e-06, | |
| "num_tokens": 960025.0, | |
| "completions/mean_length": 66.5, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 74.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.5, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 74.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.5, | |
| "kl": 0.0002476854770065984, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.64, | |
| "step": 192 | |
| }, | |
| { | |
| "loss": 2.431090479149134e-06, | |
| "grad_norm": 0.00022188037110026926, | |
| "learning_rate": 2.0000000000000003e-06, | |
| "num_tokens": 967610.0, | |
| "completions/mean_length": 94.25, | |
| "completions/min_length": 87.0, | |
| "completions/max_length": 108.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 94.25, | |
| "completions/min_terminated_length": 87.0, | |
| "completions/max_terminated_length": 108.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 94.25, | |
| "kl": 0.0024310903681907803, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6433333333333333, | |
| "step": 193 | |
| }, | |
| { | |
| "loss": 1.0377514172432711e-06, | |
| "grad_norm": 0.0001708572672214359, | |
| "learning_rate": 1.9814814814814815e-06, | |
| "num_tokens": 975168.0, | |
| "completions/mean_length": 87.5, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 99.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.5, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 99.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.5, | |
| "kl": 0.0010377514263382182, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6466666666666666, | |
| "step": 194 | |
| }, | |
| { | |
| "loss": 5.414336919784546e-05, | |
| "grad_norm": 0.8402225375175476, | |
| "learning_rate": 1.962962962962963e-06, | |
| "num_tokens": 981736.0, | |
| "completions/mean_length": 85.0, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 107.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 85.0, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 107.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 85.0, | |
| "kl": 0.05422613676637411, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.65, | |
| "step": 195 | |
| }, | |
| { | |
| "loss": 4.732796242024051e-06, | |
| "grad_norm": 0.00022060101036913693, | |
| "learning_rate": 1.944444444444445e-06, | |
| "num_tokens": 985849.0, | |
| "completions/mean_length": 69.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 79.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 69.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 79.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 69.25, | |
| "kl": 0.0047327958745881915, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6533333333333333, | |
| "step": 196 | |
| }, | |
| { | |
| "loss": 2.091497663059272e-06, | |
| "grad_norm": 0.00022720571723766625, | |
| "learning_rate": 1.925925925925926e-06, | |
| "num_tokens": 993442.0, | |
| "completions/mean_length": 96.25, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 117.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.25, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 117.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.25, | |
| "kl": 0.0020914974011247978, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6566666666666666, | |
| "step": 197 | |
| }, | |
| { | |
| "loss": 1.1918798463739222e-06, | |
| "grad_norm": 0.00013704507728107274, | |
| "learning_rate": 1.9074074074074076e-06, | |
| "num_tokens": 1001094.0, | |
| "completions/mean_length": 111.0, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 169.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 111.0, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 169.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 111.0, | |
| "kl": 0.001191879899124615, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.66, | |
| "step": 198 | |
| }, | |
| { | |
| "loss": 0.00017359084449708462, | |
| "grad_norm": 0.025992613285779953, | |
| "learning_rate": 1.888888888888889e-06, | |
| "num_tokens": 1006957.0, | |
| "completions/mean_length": 80.75, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.75, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.75, | |
| "kl": 0.17359083448536694, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6633333333333333, | |
| "step": 199 | |
| }, | |
| { | |
| "loss": 7.688890036661178e-05, | |
| "grad_norm": 0.0027668888214975595, | |
| "learning_rate": 1.8703703703703705e-06, | |
| "num_tokens": 1013479.0, | |
| "completions/mean_length": 73.5, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 84.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.5, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 84.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.5, | |
| "kl": 0.07688890118151903, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6666666666666666, | |
| "step": 200 | |
| }, | |
| { | |
| "loss": 1.7455098486607312e-06, | |
| "grad_norm": 0.0003410454955883324, | |
| "learning_rate": 1.8518518518518519e-06, | |
| "num_tokens": 1017106.0, | |
| "completions/mean_length": 82.75, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.75, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.75, | |
| "kl": 0.0017455098168284167, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.67, | |
| "step": 201 | |
| }, | |
| { | |
| "loss": 2.7156079340784345e-06, | |
| "grad_norm": 0.00021299711079336703, | |
| "learning_rate": 1.8333333333333333e-06, | |
| "num_tokens": 1020401.0, | |
| "completions/mean_length": 98.75, | |
| "completions/min_length": 90.0, | |
| "completions/max_length": 105.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 98.75, | |
| "completions/min_terminated_length": 90.0, | |
| "completions/max_terminated_length": 105.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 98.75, | |
| "kl": 0.002715607814025134, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6733333333333333, | |
| "step": 202 | |
| }, | |
| { | |
| "loss": 2.4018636395339854e-06, | |
| "grad_norm": 0.0002415847557131201, | |
| "learning_rate": 1.8148148148148151e-06, | |
| "num_tokens": 1028029.0, | |
| "completions/mean_length": 105.0, | |
| "completions/min_length": 89.0, | |
| "completions/max_length": 126.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 105.0, | |
| "completions/min_terminated_length": 89.0, | |
| "completions/max_terminated_length": 126.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 105.0, | |
| "kl": 0.0024018633412197232, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6766666666666666, | |
| "step": 203 | |
| }, | |
| { | |
| "loss": 6.398347613867372e-05, | |
| "grad_norm": 0.0005037173978053033, | |
| "learning_rate": 1.7962962962962965e-06, | |
| "num_tokens": 1034549.0, | |
| "completions/mean_length": 73.0, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 79.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.0, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 79.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.0, | |
| "kl": 0.06398347206413746, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.68, | |
| "step": 204 | |
| }, | |
| { | |
| "loss": 6.7493037931853905e-06, | |
| "grad_norm": 0.00014861871022731066, | |
| "learning_rate": 1.777777777777778e-06, | |
| "num_tokens": 1038669.0, | |
| "completions/mean_length": 68.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 68.0, | |
| "kl": 0.006749303545802832, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6833333333333333, | |
| "step": 205 | |
| }, | |
| { | |
| "loss": 6.970988124521682e-07, | |
| "grad_norm": 0.0001656811946304515, | |
| "learning_rate": 1.7592592592592594e-06, | |
| "num_tokens": 1042330.0, | |
| "completions/mean_length": 91.25, | |
| "completions/min_length": 86.0, | |
| "completions/max_length": 98.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 91.25, | |
| "completions/min_terminated_length": 86.0, | |
| "completions/max_terminated_length": 98.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 91.25, | |
| "kl": 0.0006970988397370093, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6866666666666666, | |
| "step": 206 | |
| }, | |
| { | |
| "loss": 1.9628671452665003e-06, | |
| "grad_norm": 0.00022066160454414785, | |
| "learning_rate": 1.740740740740741e-06, | |
| "num_tokens": 1049927.0, | |
| "completions/mean_length": 97.25, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 108.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 97.25, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 108.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 97.25, | |
| "kl": 0.0019628672453109175, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.69, | |
| "step": 207 | |
| }, | |
| { | |
| "loss": 4.733273613055644e-07, | |
| "grad_norm": 0.00026264035841450095, | |
| "learning_rate": 1.7222222222222224e-06, | |
| "num_tokens": 1053024.0, | |
| "completions/mean_length": 74.25, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 81.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 74.25, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 81.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 74.25, | |
| "kl": 0.0004733273417514283, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6933333333333334, | |
| "step": 208 | |
| }, | |
| { | |
| "loss": 1.3688781109522097e-05, | |
| "grad_norm": 0.00015520237502641976, | |
| "learning_rate": 1.7037037037037038e-06, | |
| "num_tokens": 1058889.0, | |
| "completions/mean_length": 81.25, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 84.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 81.25, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 84.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 81.25, | |
| "kl": 0.013688781764358282, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.6966666666666667, | |
| "step": 209 | |
| }, | |
| { | |
| "loss": 2.821790985763073e-06, | |
| "grad_norm": 0.00023798651818651706, | |
| "learning_rate": 1.6851851851851852e-06, | |
| "num_tokens": 1066430.0, | |
| "completions/mean_length": 83.25, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 86.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.25, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 86.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.25, | |
| "kl": 0.002821790927555412, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7, | |
| "step": 210 | |
| }, | |
| { | |
| "loss": 1.2744651485263603e-06, | |
| "grad_norm": 0.0001656683161854744, | |
| "learning_rate": 1.6666666666666667e-06, | |
| "num_tokens": 1074078.0, | |
| "completions/mean_length": 110.0, | |
| "completions/min_length": 93.0, | |
| "completions/max_length": 126.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 110.0, | |
| "completions/min_terminated_length": 93.0, | |
| "completions/max_terminated_length": 126.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 110.0, | |
| "kl": 0.0012744650593958795, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7033333333333334, | |
| "step": 211 | |
| }, | |
| { | |
| "loss": 3.9077349356375635e-06, | |
| "grad_norm": 0.00025481684133410454, | |
| "learning_rate": 1.648148148148148e-06, | |
| "num_tokens": 1078200.0, | |
| "completions/mean_length": 65.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 67.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 67.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.5, | |
| "kl": 0.00390773470280692, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7066666666666667, | |
| "step": 212 | |
| }, | |
| { | |
| "loss": 3.355215540068457e-06, | |
| "grad_norm": 0.00013519656204152852, | |
| "learning_rate": 1.62962962962963e-06, | |
| "num_tokens": 1082325.0, | |
| "completions/mean_length": 65.25, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.25, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.25, | |
| "kl": 0.0033552154200151563, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.71, | |
| "step": 213 | |
| }, | |
| { | |
| "loss": 1.3546480658988003e-05, | |
| "grad_norm": 0.00013723580923397094, | |
| "learning_rate": 1.6111111111111113e-06, | |
| "num_tokens": 1088185.0, | |
| "completions/mean_length": 80.0, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.0, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.0, | |
| "kl": 0.01354647846892476, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7133333333333334, | |
| "step": 214 | |
| }, | |
| { | |
| "loss": 2.0291899716085027e-07, | |
| "grad_norm": 8.23676964500919e-05, | |
| "learning_rate": 1.5925925925925927e-06, | |
| "num_tokens": 1091258.0, | |
| "completions/mean_length": 68.25, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.25, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 68.25, | |
| "kl": 0.00020291896908020135, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7166666666666667, | |
| "step": 215 | |
| }, | |
| { | |
| "loss": 2.504718281670648e-07, | |
| "grad_norm": 7.994453335413709e-05, | |
| "learning_rate": 1.5740740740740742e-06, | |
| "num_tokens": 1094994.0, | |
| "completions/mean_length": 82.0, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.0, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.0, | |
| "kl": 0.0002504718013369711, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.72, | |
| "step": 216 | |
| }, | |
| { | |
| "loss": 2.4907417355279904e-06, | |
| "grad_norm": 0.00019752308435272425, | |
| "learning_rate": 1.5555555555555558e-06, | |
| "num_tokens": 1098276.0, | |
| "completions/mean_length": 95.5, | |
| "completions/min_length": 86.0, | |
| "completions/max_length": 105.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 95.5, | |
| "completions/min_terminated_length": 86.0, | |
| "completions/max_terminated_length": 105.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 95.5, | |
| "kl": 0.002490741928340867, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7233333333333334, | |
| "step": 217 | |
| }, | |
| { | |
| "loss": 2.353185948322789e-07, | |
| "grad_norm": 4.970211739419028e-05, | |
| "learning_rate": 1.5370370370370372e-06, | |
| "num_tokens": 1102028.0, | |
| "completions/mean_length": 86.0, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 86.0, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 86.0, | |
| "kl": 0.00023531859005743172, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7266666666666667, | |
| "step": 218 | |
| }, | |
| { | |
| "loss": 6.986363587202504e-05, | |
| "grad_norm": 0.0020838803611695766, | |
| "learning_rate": 1.5185185185185186e-06, | |
| "num_tokens": 1108573.0, | |
| "completions/mean_length": 79.25, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 92.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.25, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 92.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.25, | |
| "kl": 0.06986362673342228, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.73, | |
| "step": 219 | |
| }, | |
| { | |
| "loss": 3.2729608392401133e-06, | |
| "grad_norm": 0.00012693583266809583, | |
| "learning_rate": 1.5e-06, | |
| "num_tokens": 1112713.0, | |
| "completions/mean_length": 78.0, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 81.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 78.0, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 81.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 78.0, | |
| "kl": 0.0032729606609791517, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7333333333333333, | |
| "step": 220 | |
| }, | |
| { | |
| "loss": 8.739248755773588e-07, | |
| "grad_norm": 0.0001070363141479902, | |
| "learning_rate": 1.4814814814814815e-06, | |
| "num_tokens": 1120305.0, | |
| "completions/mean_length": 96.0, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 126.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.0, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 126.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.0, | |
| "kl": 0.0008739249060454313, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7366666666666667, | |
| "step": 221 | |
| }, | |
| { | |
| "loss": 4.572212901621242e-07, | |
| "grad_norm": 0.0001162015541922301, | |
| "learning_rate": 1.4629629629629629e-06, | |
| "num_tokens": 1124046.0, | |
| "completions/mean_length": 83.25, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 92.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.25, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 92.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.25, | |
| "kl": 0.00045722126014879905, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.74, | |
| "step": 222 | |
| }, | |
| { | |
| "loss": 6.261884118430316e-05, | |
| "grad_norm": 0.0005726204835809767, | |
| "learning_rate": 1.4444444444444445e-06, | |
| "num_tokens": 1130563.0, | |
| "completions/mean_length": 72.25, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.25, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.25, | |
| "kl": 0.06261884327977896, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7433333333333333, | |
| "step": 223 | |
| }, | |
| { | |
| "loss": 4.475028981687501e-05, | |
| "grad_norm": 0.0001675942912697792, | |
| "learning_rate": 1.4259259259259261e-06, | |
| "num_tokens": 1134698.0, | |
| "completions/mean_length": 72.75, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 82.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.75, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 82.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.75, | |
| "kl": 0.04475028347223997, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7466666666666667, | |
| "step": 224 | |
| }, | |
| { | |
| "loss": 9.278413199353963e-05, | |
| "grad_norm": 0.02054877206683159, | |
| "learning_rate": 1.4074074074074075e-06, | |
| "num_tokens": 1140079.0, | |
| "completions/mean_length": 80.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.25, | |
| "kl": 0.09278413926949725, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.75, | |
| "step": 225 | |
| }, | |
| { | |
| "loss": 3.5111606848658994e-06, | |
| "grad_norm": 0.00015062233433127403, | |
| "learning_rate": 1.3888888888888892e-06, | |
| "num_tokens": 1145417.0, | |
| "completions/mean_length": 69.5, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 69.5, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 69.5, | |
| "kl": 0.0035111604956910014, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7533333333333333, | |
| "step": 226 | |
| }, | |
| { | |
| "loss": 3.048219241463812e-06, | |
| "grad_norm": 0.0002493282372597605, | |
| "learning_rate": 1.3703703703703706e-06, | |
| "num_tokens": 1152964.0, | |
| "completions/mean_length": 84.75, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 90.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 84.75, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 90.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 84.75, | |
| "kl": 0.0030482191941700876, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7566666666666667, | |
| "step": 227 | |
| }, | |
| { | |
| "loss": 4.653018186218105e-05, | |
| "grad_norm": 0.010008263401687145, | |
| "learning_rate": 1.351851851851852e-06, | |
| "num_tokens": 1158323.0, | |
| "completions/mean_length": 74.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 87.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 74.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 87.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 74.75, | |
| "kl": 0.04653018235694617, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.76, | |
| "step": 228 | |
| }, | |
| { | |
| "loss": 2.2013282432453707e-06, | |
| "grad_norm": 0.00024057507107499987, | |
| "learning_rate": 1.3333333333333334e-06, | |
| "num_tokens": 1165923.0, | |
| "completions/mean_length": 98.0, | |
| "completions/min_length": 89.0, | |
| "completions/max_length": 110.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 98.0, | |
| "completions/min_terminated_length": 89.0, | |
| "completions/max_terminated_length": 110.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 98.0, | |
| "kl": 0.0022013279667589813, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7633333333333333, | |
| "step": 229 | |
| }, | |
| { | |
| "loss": 3.400763944227947e-06, | |
| "grad_norm": 0.00014092975470703095, | |
| "learning_rate": 1.3148148148148148e-06, | |
| "num_tokens": 1171254.0, | |
| "completions/mean_length": 67.75, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.75, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.75, | |
| "kl": 0.0034007637877948582, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7666666666666667, | |
| "step": 230 | |
| }, | |
| { | |
| "loss": 2.674615643627476e-06, | |
| "grad_norm": 0.0001542935351608321, | |
| "learning_rate": 1.2962962962962962e-06, | |
| "num_tokens": 1174519.0, | |
| "completions/mean_length": 91.25, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 106.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 91.25, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 106.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 91.25, | |
| "kl": 0.0026746157091110945, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.77, | |
| "step": 231 | |
| }, | |
| { | |
| "loss": 3.4207073440484237e-06, | |
| "grad_norm": 0.0001506136468378827, | |
| "learning_rate": 1.2777777777777779e-06, | |
| "num_tokens": 1179844.0, | |
| "completions/mean_length": 66.25, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.25, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.25, | |
| "kl": 0.0034207070129923522, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7733333333333333, | |
| "step": 232 | |
| }, | |
| { | |
| "loss": 5.914364737691358e-05, | |
| "grad_norm": 0.00032679567812010646, | |
| "learning_rate": 1.2592592592592593e-06, | |
| "num_tokens": 1186375.0, | |
| "completions/mean_length": 75.75, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 80.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.75, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 80.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 75.75, | |
| "kl": 0.059143644757568836, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7766666666666666, | |
| "step": 233 | |
| }, | |
| { | |
| "loss": 6.0001366364303976e-05, | |
| "grad_norm": 0.0003789443871937692, | |
| "learning_rate": 1.240740740740741e-06, | |
| "num_tokens": 1192896.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 70.0, | |
| "completions/max_length": 76.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 70.0, | |
| "completions/max_terminated_length": 76.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 0.06000136863440275, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.78, | |
| "step": 234 | |
| }, | |
| { | |
| "loss": 3.0009293823241023e-06, | |
| "grad_norm": 8.311862620757893e-05, | |
| "learning_rate": 1.2222222222222223e-06, | |
| "num_tokens": 1197041.0, | |
| "completions/mean_length": 79.25, | |
| "completions/min_length": 70.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.25, | |
| "completions/min_terminated_length": 70.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.25, | |
| "kl": 0.0030009292531758547, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7833333333333333, | |
| "step": 235 | |
| }, | |
| { | |
| "loss": 3.014234152942663e-06, | |
| "grad_norm": 0.00035083730472251773, | |
| "learning_rate": 1.2037037037037037e-06, | |
| "num_tokens": 1200313.0, | |
| "completions/mean_length": 93.0, | |
| "completions/min_length": 91.0, | |
| "completions/max_length": 94.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.0, | |
| "completions/min_terminated_length": 91.0, | |
| "completions/max_terminated_length": 94.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.0, | |
| "kl": 0.0030142338655423373, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7866666666666666, | |
| "step": 236 | |
| }, | |
| { | |
| "loss": 0.00014745444059371948, | |
| "grad_norm": 1.219290852546692, | |
| "learning_rate": 1.1851851851851854e-06, | |
| "num_tokens": 1206842.0, | |
| "completions/mean_length": 75.25, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 80.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.25, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 80.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 75.25, | |
| "kl": 0.1474989177659154, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.79, | |
| "step": 237 | |
| }, | |
| { | |
| "loss": 3.514139962135232e-06, | |
| "grad_norm": 0.00017870671581476927, | |
| "learning_rate": 1.1666666666666668e-06, | |
| "num_tokens": 1210924.0, | |
| "completions/mean_length": 62.5, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 62.5, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 62.5, | |
| "kl": 0.0035141398548148572, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7933333333333333, | |
| "step": 238 | |
| }, | |
| { | |
| "loss": 2.2071330022299662e-05, | |
| "grad_norm": 0.00011804162204498425, | |
| "learning_rate": 1.1481481481481482e-06, | |
| "num_tokens": 1215030.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.5, | |
| "kl": 0.022071330808103085, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.7966666666666666, | |
| "step": 239 | |
| }, | |
| { | |
| "loss": 3.2811740879878926e-07, | |
| "grad_norm": 6.698595097986981e-05, | |
| "learning_rate": 1.1296296296296296e-06, | |
| "num_tokens": 1218767.0, | |
| "completions/mean_length": 82.25, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 90.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.25, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 90.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.25, | |
| "kl": 0.0003281173740106169, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8, | |
| "step": 240 | |
| }, | |
| { | |
| "loss": 8.029346645344049e-05, | |
| "grad_norm": 0.003061961382627487, | |
| "learning_rate": 1.111111111111111e-06, | |
| "num_tokens": 1225288.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 79.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 79.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 0.08029345702379942, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8033333333333333, | |
| "step": 241 | |
| }, | |
| { | |
| "loss": 3.1850831874180585e-06, | |
| "grad_norm": 0.00023321631306316704, | |
| "learning_rate": 1.0925925925925927e-06, | |
| "num_tokens": 1232842.0, | |
| "completions/mean_length": 86.5, | |
| "completions/min_length": 81.0, | |
| "completions/max_length": 93.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 86.5, | |
| "completions/min_terminated_length": 81.0, | |
| "completions/max_terminated_length": 93.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 86.5, | |
| "kl": 0.0031850830418989062, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8066666666666666, | |
| "step": 242 | |
| }, | |
| { | |
| "loss": 1.0224040352113661e-06, | |
| "grad_norm": 0.0001222416030941531, | |
| "learning_rate": 1.074074074074074e-06, | |
| "num_tokens": 1240461.0, | |
| "completions/mean_length": 102.75, | |
| "completions/min_length": 69.0, | |
| "completions/max_length": 124.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 102.75, | |
| "completions/min_terminated_length": 69.0, | |
| "completions/max_terminated_length": 124.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 102.75, | |
| "kl": 0.0010224040161119774, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.81, | |
| "step": 243 | |
| }, | |
| { | |
| "loss": 6.793439388275146e-05, | |
| "grad_norm": 1.779667615890503, | |
| "learning_rate": 1.0555555555555557e-06, | |
| "num_tokens": 1244570.0, | |
| "completions/mean_length": 68.25, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.25, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 68.25, | |
| "kl": 0.06801265408284962, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8133333333333334, | |
| "step": 244 | |
| }, | |
| { | |
| "loss": 6.777412636438385e-06, | |
| "grad_norm": 0.0001956024789251387, | |
| "learning_rate": 1.0370370370370371e-06, | |
| "num_tokens": 1248717.0, | |
| "completions/mean_length": 67.75, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 71.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.75, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 71.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.75, | |
| "kl": 0.006777412025257945, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8166666666666667, | |
| "step": 245 | |
| }, | |
| { | |
| "loss": 2.703924792513135e-06, | |
| "grad_norm": 0.0002504728618077934, | |
| "learning_rate": 1.0185185185185185e-06, | |
| "num_tokens": 1256289.0, | |
| "completions/mean_length": 91.0, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 109.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 91.0, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 109.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 91.0, | |
| "kl": 0.0027039245469495654, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.82, | |
| "step": 246 | |
| }, | |
| { | |
| "loss": 4.082328359800158e-06, | |
| "grad_norm": 0.00015146788791753352, | |
| "learning_rate": 1.0000000000000002e-06, | |
| "num_tokens": 1260392.0, | |
| "completions/mean_length": 65.75, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.75, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.75, | |
| "kl": 0.0040823284070938826, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8233333333333334, | |
| "step": 247 | |
| }, | |
| { | |
| "loss": 1.7670362240096438e-07, | |
| "grad_norm": 7.667240424780175e-05, | |
| "learning_rate": 9.814814814814816e-07, | |
| "num_tokens": 1263449.0, | |
| "completions/mean_length": 64.25, | |
| "completions/min_length": 58.0, | |
| "completions/max_length": 73.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.25, | |
| "completions/min_terminated_length": 58.0, | |
| "completions/max_terminated_length": 73.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.25, | |
| "kl": 0.00017670361557975411, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8266666666666667, | |
| "step": 248 | |
| }, | |
| { | |
| "loss": 1.796075412130449e-05, | |
| "grad_norm": 0.004890889395028353, | |
| "learning_rate": 9.62962962962963e-07, | |
| "num_tokens": 1268777.0, | |
| "completions/mean_length": 67.0, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.0, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.0, | |
| "kl": 0.017960750265046954, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.83, | |
| "step": 249 | |
| }, | |
| { | |
| "loss": 6.116179429227486e-05, | |
| "grad_norm": 0.0004232367209624499, | |
| "learning_rate": 9.444444444444445e-07, | |
| "num_tokens": 1275311.0, | |
| "completions/mean_length": 76.5, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 84.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 76.5, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 84.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 76.5, | |
| "kl": 0.061161791905760765, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8333333333333334, | |
| "step": 250 | |
| }, | |
| { | |
| "loss": 2.5561171241861302e-06, | |
| "grad_norm": 0.0001311030937358737, | |
| "learning_rate": 9.259259259259259e-07, | |
| "num_tokens": 1278597.0, | |
| "completions/mean_length": 96.5, | |
| "completions/min_length": 83.0, | |
| "completions/max_length": 107.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 96.5, | |
| "completions/min_terminated_length": 83.0, | |
| "completions/max_terminated_length": 107.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 96.5, | |
| "kl": 0.002556117222411558, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8366666666666667, | |
| "step": 251 | |
| }, | |
| { | |
| "loss": 2.788712549772754e-07, | |
| "grad_norm": 9.857428085524589e-05, | |
| "learning_rate": 9.074074074074076e-07, | |
| "num_tokens": 1281667.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 59.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 59.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.5, | |
| "kl": 0.0002788712463370757, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.84, | |
| "step": 252 | |
| }, | |
| { | |
| "loss": 0.0001861139026004821, | |
| "grad_norm": 0.02339707501232624, | |
| "learning_rate": 8.88888888888889e-07, | |
| "num_tokens": 1288204.0, | |
| "completions/mean_length": 77.25, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 84.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 77.25, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 84.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 77.25, | |
| "kl": 0.18611389677971601, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8433333333333334, | |
| "step": 253 | |
| }, | |
| { | |
| "loss": 3.3283171774201037e-07, | |
| "grad_norm": 0.00014298099267762154, | |
| "learning_rate": 8.703703703703705e-07, | |
| "num_tokens": 1291262.0, | |
| "completions/mean_length": 64.5, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 67.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.5, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 67.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.5, | |
| "kl": 0.0003328317070554476, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8466666666666667, | |
| "step": 254 | |
| }, | |
| { | |
| "loss": 7.4402680638741e-07, | |
| "grad_norm": 0.00017611995281185955, | |
| "learning_rate": 8.518518518518519e-07, | |
| "num_tokens": 1294912.0, | |
| "completions/mean_length": 88.5, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 117.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 88.5, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 117.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 88.5, | |
| "kl": 0.0007440267727361061, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.85, | |
| "step": 255 | |
| }, | |
| { | |
| "loss": 1.2963839708390879e-06, | |
| "grad_norm": 0.00015397589595522732, | |
| "learning_rate": 8.333333333333333e-07, | |
| "num_tokens": 1299136.0, | |
| "completions/mean_length": 83.0, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.0, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.0, | |
| "kl": 0.0012963838526047766, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8533333333333334, | |
| "step": 256 | |
| }, | |
| { | |
| "loss": 2.4750668671913445e-05, | |
| "grad_norm": 0.0002352478913962841, | |
| "learning_rate": 8.14814814814815e-07, | |
| "num_tokens": 1303250.0, | |
| "completions/mean_length": 67.5, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.5, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.5, | |
| "kl": 0.02475067088380456, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8566666666666667, | |
| "step": 257 | |
| }, | |
| { | |
| "loss": 3.3264154808421154e-06, | |
| "grad_norm": 0.0001363822229905054, | |
| "learning_rate": 7.962962962962964e-07, | |
| "num_tokens": 1307438.0, | |
| "completions/mean_length": 80.0, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 91.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.0, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 91.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.0, | |
| "kl": 0.003326415433548391, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.86, | |
| "step": 258 | |
| }, | |
| { | |
| "loss": 6.655222387053072e-05, | |
| "grad_norm": 0.0011064077261835337, | |
| "learning_rate": 7.777777777777779e-07, | |
| "num_tokens": 1313966.0, | |
| "completions/mean_length": 75.0, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 81.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.0, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 81.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 75.0, | |
| "kl": 0.06655221618711948, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8633333333333333, | |
| "step": 259 | |
| }, | |
| { | |
| "loss": 2.816607036493224e-07, | |
| "grad_norm": 6.245569966267794e-05, | |
| "learning_rate": 7.592592592592593e-07, | |
| "num_tokens": 1317725.0, | |
| "completions/mean_length": 87.75, | |
| "completions/min_length": 83.0, | |
| "completions/max_length": 99.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.75, | |
| "completions/min_terminated_length": 83.0, | |
| "completions/max_terminated_length": 99.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.75, | |
| "kl": 0.00028166069932922255, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8666666666666667, | |
| "step": 260 | |
| }, | |
| { | |
| "loss": 9.876027434074786e-06, | |
| "grad_norm": 0.00015478942077606916, | |
| "learning_rate": 7.407407407407407e-07, | |
| "num_tokens": 1321854.0, | |
| "completions/mean_length": 66.25, | |
| "completions/min_length": 61.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.25, | |
| "completions/min_terminated_length": 61.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.25, | |
| "kl": 0.009876027470454574, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.87, | |
| "step": 261 | |
| }, | |
| { | |
| "loss": 1.2832955690100789e-05, | |
| "grad_norm": 7.08275765646249e-05, | |
| "learning_rate": 7.222222222222222e-07, | |
| "num_tokens": 1327722.0, | |
| "completions/mean_length": 82.0, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 86.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 82.0, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 86.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 82.0, | |
| "kl": 0.012832952430471778, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8733333333333333, | |
| "step": 262 | |
| }, | |
| { | |
| "loss": 7.638001875420741e-07, | |
| "grad_norm": 8.452033216599375e-05, | |
| "learning_rate": 7.037037037037038e-07, | |
| "num_tokens": 1335304.0, | |
| "completions/mean_length": 93.5, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 123.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.5, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 123.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.5, | |
| "kl": 0.0007638001334271394, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8766666666666667, | |
| "step": 263 | |
| }, | |
| { | |
| "loss": 7.4834065344475675e-06, | |
| "grad_norm": 0.00014483257837127894, | |
| "learning_rate": 6.851851851851853e-07, | |
| "num_tokens": 1339449.0, | |
| "completions/mean_length": 72.25, | |
| "completions/min_length": 68.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.25, | |
| "completions/min_terminated_length": 68.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.25, | |
| "kl": 0.007483406458050013, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.88, | |
| "step": 264 | |
| }, | |
| { | |
| "loss": 1.3292547009768896e-05, | |
| "grad_norm": 7.537764759035781e-05, | |
| "learning_rate": 6.666666666666667e-07, | |
| "num_tokens": 1345307.0, | |
| "completions/mean_length": 79.5, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.5, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.5, | |
| "kl": 0.013292544987052679, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8833333333333333, | |
| "step": 265 | |
| }, | |
| { | |
| "loss": 1.3370675333135296e-05, | |
| "grad_norm": 0.00010329321230528876, | |
| "learning_rate": 6.481481481481481e-07, | |
| "num_tokens": 1351168.0, | |
| "completions/mean_length": 80.25, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 85.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.25, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 85.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.25, | |
| "kl": 0.013370674336329103, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8866666666666667, | |
| "step": 266 | |
| }, | |
| { | |
| "loss": 3.860032848024275e-06, | |
| "grad_norm": 0.00016948120901361108, | |
| "learning_rate": 6.296296296296296e-07, | |
| "num_tokens": 1355305.0, | |
| "completions/mean_length": 65.25, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.25, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.25, | |
| "kl": 0.0038600328844040632, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.89, | |
| "step": 267 | |
| }, | |
| { | |
| "loss": 6.24434178462252e-05, | |
| "grad_norm": 0.00035368569660931826, | |
| "learning_rate": 6.111111111111112e-07, | |
| "num_tokens": 1361826.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 70.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 70.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 0.06244341749697924, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8933333333333333, | |
| "step": 268 | |
| }, | |
| { | |
| "loss": 5.83292858209461e-05, | |
| "grad_norm": 0.0003988529497291893, | |
| "learning_rate": 5.925925925925927e-07, | |
| "num_tokens": 1368375.0, | |
| "completions/mean_length": 80.25, | |
| "completions/min_length": 74.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 80.25, | |
| "completions/min_terminated_length": 74.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 80.25, | |
| "kl": 0.058329286985099316, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.8966666666666666, | |
| "step": 269 | |
| }, | |
| { | |
| "loss": 1.5467405319213867e-05, | |
| "grad_norm": 0.8754695057868958, | |
| "learning_rate": 5.740740740740741e-07, | |
| "num_tokens": 1374220.0, | |
| "completions/mean_length": 76.25, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 76.25, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 0.824999988079071, | |
| "rewards/reward_fn/std": 0.3499999940395355, | |
| "reward": 0.824999988079071, | |
| "reward_std": 0.3499999940395355, | |
| "frac_reward_zero_std": 0.0, | |
| "completion_length": 76.25, | |
| "kl": 0.015492761274799705, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9, | |
| "step": 270 | |
| }, | |
| { | |
| "loss": 9.359965042676777e-05, | |
| "grad_norm": 0.004699740558862686, | |
| "learning_rate": 5.555555555555555e-07, | |
| "num_tokens": 1380744.0, | |
| "completions/mean_length": 74.0, | |
| "completions/min_length": 70.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 74.0, | |
| "completions/min_terminated_length": 70.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 74.0, | |
| "kl": 0.0935996463522315, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9033333333333333, | |
| "step": 271 | |
| }, | |
| { | |
| "loss": 1.334827538812533e-05, | |
| "grad_norm": 0.00010597281652735546, | |
| "learning_rate": 5.37037037037037e-07, | |
| "num_tokens": 1386603.0, | |
| "completions/mean_length": 79.75, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 79.75, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 79.75, | |
| "kl": 0.013348274864256382, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9066666666666666, | |
| "step": 272 | |
| }, | |
| { | |
| "loss": 2.0391826183185913e-06, | |
| "grad_norm": 0.00010451644629938528, | |
| "learning_rate": 5.185185185185186e-07, | |
| "num_tokens": 1389896.0, | |
| "completions/mean_length": 98.25, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 108.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 98.25, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 108.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 98.25, | |
| "kl": 0.0020391825237311423, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.91, | |
| "step": 273 | |
| }, | |
| { | |
| "loss": 2.5969702619477175e-05, | |
| "grad_norm": 0.002803744515404105, | |
| "learning_rate": 5.000000000000001e-07, | |
| "num_tokens": 1397446.0, | |
| "completions/mean_length": 85.5, | |
| "completions/min_length": 73.0, | |
| "completions/max_length": 95.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 85.5, | |
| "completions/min_terminated_length": 73.0, | |
| "completions/max_terminated_length": 95.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 85.5, | |
| "kl": 0.02596970333252102, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9133333333333333, | |
| "step": 274 | |
| }, | |
| { | |
| "loss": 4.015822014480364e-06, | |
| "grad_norm": 0.00020393294107634574, | |
| "learning_rate": 4.814814814814815e-07, | |
| "num_tokens": 1402799.0, | |
| "completions/mean_length": 73.25, | |
| "completions/min_length": 71.0, | |
| "completions/max_length": 75.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.25, | |
| "completions/min_terminated_length": 71.0, | |
| "completions/max_terminated_length": 75.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.25, | |
| "kl": 0.004015822021756321, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9166666666666666, | |
| "step": 275 | |
| }, | |
| { | |
| "loss": 8.066627970038098e-07, | |
| "grad_norm": 0.0001106026757042855, | |
| "learning_rate": 4.6296296296296297e-07, | |
| "num_tokens": 1410382.0, | |
| "completions/mean_length": 93.75, | |
| "completions/min_length": 68.0, | |
| "completions/max_length": 108.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.75, | |
| "completions/min_terminated_length": 68.0, | |
| "completions/max_terminated_length": 108.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.75, | |
| "kl": 0.0008066627269727178, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.92, | |
| "step": 276 | |
| }, | |
| { | |
| "loss": 1.6011627224088443e-07, | |
| "grad_norm": 0.00010368180664954707, | |
| "learning_rate": 4.444444444444445e-07, | |
| "num_tokens": 1413431.0, | |
| "completions/mean_length": 62.25, | |
| "completions/min_length": 57.0, | |
| "completions/max_length": 67.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 62.25, | |
| "completions/min_terminated_length": 57.0, | |
| "completions/max_terminated_length": 67.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 62.25, | |
| "kl": 0.00016011625484679826, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9233333333333333, | |
| "step": 277 | |
| }, | |
| { | |
| "loss": 3.3608453122724313e-06, | |
| "grad_norm": 0.00012059447908541188, | |
| "learning_rate": 4.2592592592592596e-07, | |
| "num_tokens": 1418764.0, | |
| "completions/mean_length": 68.25, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 72.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 68.25, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 72.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 68.25, | |
| "kl": 0.0033608448575250804, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9266666666666666, | |
| "step": 278 | |
| }, | |
| { | |
| "loss": 1.3088694686302915e-05, | |
| "grad_norm": 0.0001294590183533728, | |
| "learning_rate": 4.074074074074075e-07, | |
| "num_tokens": 1424639.0, | |
| "completions/mean_length": 83.75, | |
| "completions/min_length": 75.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 83.75, | |
| "completions/min_terminated_length": 75.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 83.75, | |
| "kl": 0.013088694540783763, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.93, | |
| "step": 279 | |
| }, | |
| { | |
| "loss": 3.0727476314496016e-06, | |
| "grad_norm": 0.00021840138651896268, | |
| "learning_rate": 3.8888888888888895e-07, | |
| "num_tokens": 1432245.0, | |
| "completions/mean_length": 99.5, | |
| "completions/min_length": 79.0, | |
| "completions/max_length": 150.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 99.5, | |
| "completions/min_terminated_length": 79.0, | |
| "completions/max_terminated_length": 150.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 99.5, | |
| "kl": 0.003072747669648379, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9333333333333333, | |
| "step": 280 | |
| }, | |
| { | |
| "loss": 3.3646563224465353e-07, | |
| "grad_norm": 7.528693822678179e-05, | |
| "learning_rate": 3.7037037037037036e-07, | |
| "num_tokens": 1436014.0, | |
| "completions/mean_length": 90.25, | |
| "completions/min_length": 82.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.25, | |
| "completions/min_terminated_length": 82.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 90.25, | |
| "kl": 0.00033646558586042374, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9366666666666666, | |
| "step": 281 | |
| }, | |
| { | |
| "loss": 4.197281668893993e-06, | |
| "grad_norm": 0.0001576204813318327, | |
| "learning_rate": 3.518518518518519e-07, | |
| "num_tokens": 1440141.0, | |
| "completions/mean_length": 64.75, | |
| "completions/min_length": 63.0, | |
| "completions/max_length": 68.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 64.75, | |
| "completions/min_terminated_length": 63.0, | |
| "completions/max_terminated_length": 68.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 64.75, | |
| "kl": 0.004197281436063349, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.94, | |
| "step": 282 | |
| }, | |
| { | |
| "loss": 6.141619815025479e-05, | |
| "grad_norm": 0.0003577295283321291, | |
| "learning_rate": 3.3333333333333335e-07, | |
| "num_tokens": 1446654.0, | |
| "completions/mean_length": 71.25, | |
| "completions/min_length": 66.0, | |
| "completions/max_length": 77.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.25, | |
| "completions/min_terminated_length": 66.0, | |
| "completions/max_terminated_length": 77.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.25, | |
| "kl": 0.06141619384288788, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9433333333333334, | |
| "step": 283 | |
| }, | |
| { | |
| "loss": 6.073229883440945e-07, | |
| "grad_norm": 0.00013481290079653263, | |
| "learning_rate": 3.148148148148148e-07, | |
| "num_tokens": 1450275.0, | |
| "completions/mean_length": 81.25, | |
| "completions/min_length": 72.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 81.25, | |
| "completions/min_terminated_length": 72.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 81.25, | |
| "kl": 0.0006073229742469266, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9466666666666667, | |
| "step": 284 | |
| }, | |
| { | |
| "loss": 3.8584028061450226e-07, | |
| "grad_norm": 8.886020805221051e-05, | |
| "learning_rate": 2.9629629629629634e-07, | |
| "num_tokens": 1454030.0, | |
| "completions/mean_length": 86.75, | |
| "completions/min_length": 80.0, | |
| "completions/max_length": 94.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 86.75, | |
| "completions/min_terminated_length": 80.0, | |
| "completions/max_terminated_length": 94.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 86.75, | |
| "kl": 0.00038584022695431486, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.95, | |
| "step": 285 | |
| }, | |
| { | |
| "loss": 8.194695692509413e-05, | |
| "grad_norm": 0.006449210457503796, | |
| "learning_rate": 2.7777777777777776e-07, | |
| "num_tokens": 1460549.0, | |
| "completions/mean_length": 72.75, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 76.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 72.75, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 76.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 72.75, | |
| "kl": 0.08194695319980383, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9533333333333334, | |
| "step": 286 | |
| }, | |
| { | |
| "loss": 3.253871227570926e-06, | |
| "grad_norm": 0.0002493150532245636, | |
| "learning_rate": 2.592592592592593e-07, | |
| "num_tokens": 1468154.0, | |
| "completions/mean_length": 99.25, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 133.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 99.25, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 133.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 99.25, | |
| "kl": 0.0032538710220251232, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9566666666666667, | |
| "step": 287 | |
| }, | |
| { | |
| "loss": 1.0019080036727246e-06, | |
| "grad_norm": 0.00011101938434876502, | |
| "learning_rate": 2.4074074074074075e-07, | |
| "num_tokens": 1475772.0, | |
| "completions/mean_length": 102.5, | |
| "completions/min_length": 94.0, | |
| "completions/max_length": 121.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 102.5, | |
| "completions/min_terminated_length": 94.0, | |
| "completions/max_terminated_length": 121.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 102.5, | |
| "kl": 0.0010019079127232544, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.96, | |
| "step": 288 | |
| }, | |
| { | |
| "loss": 6.422043952625245e-05, | |
| "grad_norm": 0.00039836866199038923, | |
| "learning_rate": 2.2222222222222224e-07, | |
| "num_tokens": 1482286.0, | |
| "completions/mean_length": 71.5, | |
| "completions/min_length": 67.0, | |
| "completions/max_length": 83.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 71.5, | |
| "completions/min_terminated_length": 67.0, | |
| "completions/max_terminated_length": 83.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 71.5, | |
| "kl": 0.06422043964266777, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9633333333333334, | |
| "step": 289 | |
| }, | |
| { | |
| "loss": 6.665252385573694e-07, | |
| "grad_norm": 0.0001485645625507459, | |
| "learning_rate": 2.0370370370370374e-07, | |
| "num_tokens": 1485944.0, | |
| "completions/mean_length": 90.5, | |
| "completions/min_length": 76.0, | |
| "completions/max_length": 114.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 90.5, | |
| "completions/min_terminated_length": 76.0, | |
| "completions/max_terminated_length": 114.0, | |
| "rewards/reward_fn/mean": 2.4000000953674316, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.4000000953674316, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 90.5, | |
| "kl": 0.0006665251858066767, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9666666666666667, | |
| "step": 290 | |
| }, | |
| { | |
| "loss": 3.030107109225355e-06, | |
| "grad_norm": 0.00021772447507828474, | |
| "learning_rate": 1.8518518518518518e-07, | |
| "num_tokens": 1493511.0, | |
| "completions/mean_length": 89.75, | |
| "completions/min_length": 85.0, | |
| "completions/max_length": 93.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 89.75, | |
| "completions/min_terminated_length": 85.0, | |
| "completions/max_terminated_length": 93.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 89.75, | |
| "kl": 0.0030301069491542876, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.97, | |
| "step": 291 | |
| }, | |
| { | |
| "loss": 6.297694199020043e-05, | |
| "grad_norm": 0.000519341672770679, | |
| "learning_rate": 1.6666666666666668e-07, | |
| "num_tokens": 1500034.0, | |
| "completions/mean_length": 73.75, | |
| "completions/min_length": 68.0, | |
| "completions/max_length": 84.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 73.75, | |
| "completions/min_terminated_length": 68.0, | |
| "completions/max_terminated_length": 84.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 73.75, | |
| "kl": 0.06297693960368633, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9733333333333334, | |
| "step": 292 | |
| }, | |
| { | |
| "loss": 7.2916241151688155e-06, | |
| "grad_norm": 0.000130257525597699, | |
| "learning_rate": 1.4814814814814817e-07, | |
| "num_tokens": 1504157.0, | |
| "completions/mean_length": 65.75, | |
| "completions/min_length": 62.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 65.75, | |
| "completions/min_terminated_length": 62.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 65.75, | |
| "kl": 0.007291623973287642, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9766666666666667, | |
| "step": 293 | |
| }, | |
| { | |
| "loss": 3.9262754398805555e-06, | |
| "grad_norm": 0.00018200626072939485, | |
| "learning_rate": 1.2962962962962964e-07, | |
| "num_tokens": 1509486.0, | |
| "completions/mean_length": 67.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 70.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 67.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 70.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 67.25, | |
| "kl": 0.003926275297999382, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.98, | |
| "step": 294 | |
| }, | |
| { | |
| "loss": 1.8765720710689493e-07, | |
| "grad_norm": 5.156537008588202e-05, | |
| "learning_rate": 1.1111111111111112e-07, | |
| "num_tokens": 1513231.0, | |
| "completions/mean_length": 84.25, | |
| "completions/min_length": 77.0, | |
| "completions/max_length": 88.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 84.25, | |
| "completions/min_terminated_length": 77.0, | |
| "completions/max_terminated_length": 88.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 84.25, | |
| "kl": 0.00018765719505609013, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9833333333333333, | |
| "step": 295 | |
| }, | |
| { | |
| "loss": 3.550860867562733e-07, | |
| "grad_norm": 0.00021035087411291897, | |
| "learning_rate": 9.259259259259259e-08, | |
| "num_tokens": 1516331.0, | |
| "completions/mean_length": 75.0, | |
| "completions/min_length": 64.0, | |
| "completions/max_length": 80.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 75.0, | |
| "completions/min_terminated_length": 64.0, | |
| "completions/max_terminated_length": 80.0, | |
| "rewards/reward_fn/mean": 1.600000023841858, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.600000023841858, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 75.0, | |
| "kl": 0.0003550860710674897, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9866666666666667, | |
| "step": 296 | |
| }, | |
| { | |
| "loss": 3.470086812740192e-05, | |
| "grad_norm": 0.010448559187352657, | |
| "learning_rate": 7.407407407407409e-08, | |
| "num_tokens": 1521656.0, | |
| "completions/mean_length": 66.25, | |
| "completions/min_length": 65.0, | |
| "completions/max_length": 69.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 66.25, | |
| "completions/min_terminated_length": 65.0, | |
| "completions/max_terminated_length": 69.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 66.25, | |
| "kl": 0.034700864111073315, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.99, | |
| "step": 297 | |
| }, | |
| { | |
| "loss": 1.6504607629030943e-05, | |
| "grad_norm": 0.0001237893447978422, | |
| "learning_rate": 5.555555555555556e-08, | |
| "num_tokens": 1525780.0, | |
| "completions/mean_length": 70.0, | |
| "completions/min_length": 60.0, | |
| "completions/max_length": 78.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 70.0, | |
| "completions/min_terminated_length": 60.0, | |
| "completions/max_terminated_length": 78.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 70.0, | |
| "kl": 0.016504606464877725, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9933333333333333, | |
| "step": 298 | |
| }, | |
| { | |
| "loss": 9.070388955478847e-07, | |
| "grad_norm": 0.00011076599912485108, | |
| "learning_rate": 3.703703703703704e-08, | |
| "num_tokens": 1533337.0, | |
| "completions/mean_length": 87.25, | |
| "completions/min_length": 78.0, | |
| "completions/max_length": 102.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 87.25, | |
| "completions/min_terminated_length": 78.0, | |
| "completions/max_terminated_length": 102.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 87.25, | |
| "kl": 0.0009070388769032434, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 0.9966666666666667, | |
| "step": 299 | |
| }, | |
| { | |
| "loss": 3.1342110560217407e-06, | |
| "grad_norm": 0.0003570808039512485, | |
| "learning_rate": 1.851851851851852e-08, | |
| "num_tokens": 1536612.0, | |
| "completions/mean_length": 93.75, | |
| "completions/min_length": 84.0, | |
| "completions/max_length": 100.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/mean_terminated_length": 93.75, | |
| "completions/min_terminated_length": 84.0, | |
| "completions/max_terminated_length": 100.0, | |
| "rewards/reward_fn/mean": 2.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "reward": 2.0, | |
| "reward_std": 0.0, | |
| "frac_reward_zero_std": 1.0, | |
| "completion_length": 93.75, | |
| "kl": 0.003134210826829076, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "epoch": 1.0, | |
| "step": 300 | |
| }, | |
| { | |
| "train_runtime": 6974.9196, | |
| "train_samples_per_second": 0.172, | |
| "train_steps_per_second": 0.043, | |
| "total_flos": 0.0, | |
| "train_loss": 1.3829677314691757e-05, | |
| "epoch": 1.0, | |
| "step": 300 | |
| } | |
| ] |