| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 4.512820512820513, |
| "eval_steps": 88.0, |
| "global_step": 528, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 197.0, |
| "completions/mean_length": 129.20834350585938, |
| "completions/min_length": 97.0, |
| "epoch": 0.008547008547008548, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.534827590655408, |
| "learning_rate": 4.545454545454545e-08, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 1, |
| "step_time": 32.52706161397509 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 166.0, |
| "completions/mean_length": 119.91667175292969, |
| "completions/min_length": 93.0, |
| "epoch": 0.017094017094017096, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.09090909090909e-08, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 2, |
| "step_time": 17.169726474909112 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 265.0, |
| "completions/mean_length": 134.9166717529297, |
| "completions/min_length": 90.0, |
| "epoch": 0.02564102564102564, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 5.557200661083816, |
| "learning_rate": 1.3636363636363635e-07, |
| "loss": -2.4835269840650653e-08, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.45032864809036255, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 3, |
| "step_time": 18.05642723897472 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 172.0, |
| "completions/mean_length": 114.5, |
| "completions/min_length": 85.0, |
| "epoch": 0.03418803418803419, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.033506763050039, |
| "learning_rate": 1.818181818181818e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 4, |
| "step_time": 17.040023577865213 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 178.20834350585938, |
| "completions/min_length": 91.0, |
| "epoch": 0.042735042735042736, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.2727272727272726e-07, |
| "loss": 0.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 5, |
| "step_time": 19.25745597295463 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 374.0, |
| "completions/mean_length": 146.7916717529297, |
| "completions/min_length": 82.0, |
| "epoch": 0.05128205128205128, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.4784995113475965, |
| "learning_rate": 2.727272727272727e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.4082186818122864, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 6, |
| "step_time": 19.92464966280386 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 193.0, |
| "completions/mean_length": 118.20833587646484, |
| "completions/min_length": 86.0, |
| "epoch": 0.05982905982905983, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.1818181818181815e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 7, |
| "step_time": 17.232735878089443 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 171.0, |
| "completions/mean_length": 118.58333587646484, |
| "completions/min_length": 78.0, |
| "epoch": 0.06837606837606838, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.636363636363636e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 8, |
| "step_time": 17.032443220959976 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 170.0, |
| "completions/mean_length": 105.20833587646484, |
| "completions/min_length": 74.0, |
| "epoch": 0.07692307692307693, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.090909090909091e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 9, |
| "step_time": 17.115436974912882 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 225.0, |
| "completions/mean_length": 132.2916717529297, |
| "completions/min_length": 75.0, |
| "epoch": 0.08547008547008547, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.545454545454545e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 10, |
| "step_time": 17.83178498595953 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 383.0, |
| "completions/mean_length": 132.9166717529297, |
| "completions/min_length": 79.0, |
| "epoch": 0.09401709401709402, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 11, |
| "step_time": 22.001850640866905 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 149.33334350585938, |
| "completions/min_length": 84.0, |
| "epoch": 0.10256410256410256, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.088828996571324, |
| "learning_rate": 5.454545454545454e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 12, |
| "step_time": 20.435253034811467 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 267.0, |
| "completions/mean_length": 131.7916717529297, |
| "completions/min_length": 80.0, |
| "epoch": 0.1111111111111111, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.6159068731436865, |
| "learning_rate": 5.909090909090909e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 13, |
| "step_time": 19.18023474002257 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 242.0, |
| "completions/mean_length": 140.125, |
| "completions/min_length": 107.0, |
| "epoch": 0.11965811965811966, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5370922833138883, |
| "learning_rate": 6.363636363636363e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 14, |
| "step_time": 18.721028126077726 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 153.0, |
| "completions/mean_length": 114.79167175292969, |
| "completions/min_length": 73.0, |
| "epoch": 0.1282051282051282, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.818181818181817e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 15, |
| "step_time": 18.217683811904863 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 177.0, |
| "completions/mean_length": 118.70833587646484, |
| "completions/min_length": 71.0, |
| "epoch": 0.13675213675213677, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.626448809680626, |
| "learning_rate": 7.272727272727272e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 16, |
| "step_time": 18.719733536010608 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 202.0, |
| "completions/mean_length": 127.41667175292969, |
| "completions/min_length": 80.0, |
| "epoch": 0.1452991452991453, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.327155983194954, |
| "learning_rate": 7.727272727272727e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.75, |
| "reward_std": 0.34503278136253357, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 17, |
| "step_time": 18.788225760916248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 160.0, |
| "completions/mean_length": 106.58333587646484, |
| "completions/min_length": 68.0, |
| "epoch": 0.15384615384615385, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.3260737826510676, |
| "learning_rate": 8.181818181818182e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 18, |
| "step_time": 19.161441093077883 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 191.875, |
| "completions/min_length": 82.0, |
| "epoch": 0.1623931623931624, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.63214810080271, |
| "learning_rate": 8.636363636363636e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 19, |
| "step_time": 25.229938219999894 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 212.0, |
| "completions/mean_length": 130.9166717529297, |
| "completions/min_length": 71.0, |
| "epoch": 0.17094017094017094, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.233916721417803, |
| "learning_rate": 9.09090909090909e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 20, |
| "step_time": 18.034462973009795 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 218.0, |
| "completions/mean_length": 124.70833587646484, |
| "completions/min_length": 77.0, |
| "epoch": 0.1794871794871795, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.545454545454546e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 21, |
| "step_time": 18.497448878129944 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 189.0, |
| "completions/mean_length": 119.16667175292969, |
| "completions/min_length": 81.0, |
| "epoch": 0.18803418803418803, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7486312822271826, |
| "learning_rate": 1e-06, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 22, |
| "step_time": 18.370401091873646 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 193.0, |
| "completions/mean_length": 121.875, |
| "completions/min_length": 73.0, |
| "epoch": 0.19658119658119658, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 4.167938002358859, |
| "learning_rate": 9.999946951850749e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 23, |
| "step_time": 18.35488649085164 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 211.0, |
| "completions/mean_length": 113.33333587646484, |
| "completions/min_length": 82.0, |
| "epoch": 0.20512820512820512, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.999787808528638e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 24, |
| "step_time": 17.982405610848218 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 134.2916717529297, |
| "completions/min_length": 73.0, |
| "epoch": 0.21367521367521367, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.999522573410568e-07, |
| "loss": 0.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 25, |
| "step_time": 22.952001517871395 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 172.83334350585938, |
| "completions/min_length": 117.0, |
| "epoch": 0.2222222222222222, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.056469009033167, |
| "learning_rate": 9.999151252124639e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 26, |
| "step_time": 19.86690955911763 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 143.08334350585938, |
| "completions/min_length": 87.0, |
| "epoch": 0.23076923076923078, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.667058796146634, |
| "learning_rate": 9.998673852550007e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 27, |
| "step_time": 18.870957584120333 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 390.0, |
| "completions/mean_length": 140.83334350585938, |
| "completions/min_length": 87.0, |
| "epoch": 0.23931623931623933, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.770582029033055, |
| "learning_rate": 9.998090384816739e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 28, |
| "step_time": 21.347367404960096 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 188.0, |
| "completions/mean_length": 119.0, |
| "completions/min_length": 82.0, |
| "epoch": 0.24786324786324787, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.99740086130559e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 29, |
| "step_time": 18.303538847947493 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 282.0, |
| "completions/mean_length": 170.08334350585938, |
| "completions/min_length": 113.0, |
| "epoch": 0.2564102564102564, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.762648787954098, |
| "learning_rate": 9.996605296647735e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.47419947385787964, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 30, |
| "step_time": 20.25631041289307 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 194.0, |
| "completions/mean_length": 102.125, |
| "completions/min_length": 77.0, |
| "epoch": 0.26495726495726496, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.245434575503011, |
| "learning_rate": 9.995703707724473e-07, |
| "loss": 7.450580596923828e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 31, |
| "step_time": 17.809944469947368 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 189.0, |
| "completions/mean_length": 133.7916717529297, |
| "completions/min_length": 86.0, |
| "epoch": 0.27350427350427353, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.491009315872147, |
| "learning_rate": 9.99469611366685e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 32, |
| "step_time": 16.91757989116013 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 369.0, |
| "completions/mean_length": 145.45834350585938, |
| "completions/min_length": 88.0, |
| "epoch": 0.28205128205128205, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.6390878945298266, |
| "learning_rate": 9.993582535855263e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.25, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.25, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 33, |
| "step_time": 23.588147747097537 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 157.0, |
| "completions/mean_length": 117.58333587646484, |
| "completions/min_length": 76.0, |
| "epoch": 0.2905982905982906, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.992362997919014e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 34, |
| "step_time": 15.234605289064348 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 191.0, |
| "completions/mean_length": 129.125, |
| "completions/min_length": 90.0, |
| "epoch": 0.29914529914529914, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.3070829516732703, |
| "learning_rate": 9.991037525735793e-07, |
| "loss": 0.0, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 35, |
| "step_time": 18.920867854962125 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 365.0, |
| "completions/mean_length": 153.25, |
| "completions/min_length": 81.0, |
| "epoch": 0.3076923076923077, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5248682191289955, |
| "learning_rate": 9.989606147431138e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 36, |
| "step_time": 16.305496397195384 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 194.0, |
| "completions/mean_length": 113.08333587646484, |
| "completions/min_length": 68.0, |
| "epoch": 0.3162393162393162, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.04210754137367, |
| "learning_rate": 9.988068893377838e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 37, |
| "step_time": 14.631900979904458 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 261.0, |
| "completions/mean_length": 124.58333587646484, |
| "completions/min_length": 82.0, |
| "epoch": 0.3247863247863248, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.334390951681701, |
| "learning_rate": 9.986425796195286e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 38, |
| "step_time": 19.27327976608649 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 138.0, |
| "completions/mean_length": 100.75, |
| "completions/min_length": 73.0, |
| "epoch": 0.3333333333333333, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.459942130039958, |
| "learning_rate": 9.984676890748786e-07, |
| "loss": 0.0, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 39, |
| "step_time": 18.118780083954334 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 247.0, |
| "completions/mean_length": 137.375, |
| "completions/min_length": 90.0, |
| "epoch": 0.3418803418803419, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.081520849332089, |
| "learning_rate": 9.982822214148818e-07, |
| "loss": -2.2351741790771484e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 40, |
| "step_time": 18.242434923071414 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 365.0, |
| "completions/mean_length": 152.375, |
| "completions/min_length": 82.0, |
| "epoch": 0.3504273504273504, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.3791144883269295, |
| "learning_rate": 9.98086180575025e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 41, |
| "step_time": 20.045204166090116 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/mean_length": 121.54167175292969, |
| "completions/min_length": 67.0, |
| "epoch": 0.358974358974359, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 5.245532184083511, |
| "learning_rate": 9.97879570715149e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.3900056481361389, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 42, |
| "step_time": 18.285045387223363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 158.5, |
| "completions/min_length": 96.0, |
| "epoch": 0.36752136752136755, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.976623962193626e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 43, |
| "step_time": 20.72015670221299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 258.0, |
| "completions/mean_length": 137.25, |
| "completions/min_length": 100.0, |
| "epoch": 0.37606837606837606, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.1486681012071207, |
| "learning_rate": 9.974346616959475e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.0416666679084301, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.0416666679084301, |
| "rewards/DirectReward/std": 0.20412413775920868, |
| "step": 44, |
| "step_time": 19.591946522938088 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 159.0, |
| "completions/mean_length": 125.91667175292969, |
| "completions/min_length": 88.0, |
| "epoch": 0.38461538461538464, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.97196371977262e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 45, |
| "step_time": 16.726943804882467 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 265.0, |
| "completions/mean_length": 138.6666717529297, |
| "completions/min_length": 78.0, |
| "epoch": 0.39316239316239315, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2994257220086705, |
| "learning_rate": 9.969475321196372e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 46, |
| "step_time": 19.276938800001517 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 167.0, |
| "completions/mean_length": 122.75, |
| "completions/min_length": 78.0, |
| "epoch": 0.4017094017094017, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.25264329170731, |
| "learning_rate": 9.96688147403271e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 47, |
| "step_time": 16.447938879020512 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 140.0, |
| "completions/mean_length": 110.16667175292969, |
| "completions/min_length": 77.0, |
| "epoch": 0.41025641025641024, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.964182233321149e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 48, |
| "step_time": 17.38712235307321 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 227.0, |
| "completions/mean_length": 114.54167175292969, |
| "completions/min_length": 77.0, |
| "epoch": 0.4188034188034188, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.961377656337576e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 49, |
| "step_time": 15.412687954027206 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 190.0, |
| "completions/mean_length": 124.58333587646484, |
| "completions/min_length": 95.0, |
| "epoch": 0.42735042735042733, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.958467802593045e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 50, |
| "step_time": 18.36663561104797 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 257.0, |
| "completions/mean_length": 130.95834350585938, |
| "completions/min_length": 77.0, |
| "epoch": 0.4358974358974359, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2394791008112684, |
| "learning_rate": 9.955452733832492e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 51, |
| "step_time": 16.104309950955212 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 203.20834350585938, |
| "completions/min_length": 96.0, |
| "epoch": 0.4444444444444444, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.808808998782832, |
| "learning_rate": 9.952332514033446e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 52, |
| "step_time": 21.082042321097106 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 180.0, |
| "completions/mean_length": 120.125, |
| "completions/min_length": 76.0, |
| "epoch": 0.452991452991453, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.148910279443955, |
| "learning_rate": 9.949107209404663e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.35634833574295044, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 53, |
| "step_time": 15.717554655158892 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 164.0, |
| "completions/mean_length": 110.41667175292969, |
| "completions/min_length": 83.0, |
| "epoch": 0.46153846153846156, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.94577688838472e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 54, |
| "step_time": 19.721732832025737 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 154.0, |
| "completions/mean_length": 108.125, |
| "completions/min_length": 79.0, |
| "epoch": 0.4700854700854701, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8402651772415175, |
| "learning_rate": 9.942341621640557e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 55, |
| "step_time": 17.811323655070737 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 150.0, |
| "completions/min_length": 78.0, |
| "epoch": 0.47863247863247865, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.325626684991881, |
| "learning_rate": 9.938801482065997e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 56, |
| "step_time": 22.84236340993084 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 246.0, |
| "completions/mean_length": 148.0416717529297, |
| "completions/min_length": 99.0, |
| "epoch": 0.48717948717948717, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.347657910242531, |
| "learning_rate": 9.93515654478018e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 57, |
| "step_time": 19.579915241105482 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 282.0, |
| "completions/mean_length": 141.83334350585938, |
| "completions/min_length": 94.0, |
| "epoch": 0.49572649572649574, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1592017999300435, |
| "learning_rate": 9.931406887125979e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 58, |
| "step_time": 20.130078543908894 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 213.0, |
| "completions/mean_length": 116.79167175292969, |
| "completions/min_length": 82.0, |
| "epoch": 0.5042735042735043, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.48425884241891, |
| "learning_rate": 9.92755258866835e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 59, |
| "step_time": 18.937865917105228 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 467.0, |
| "completions/mean_length": 132.9166717529297, |
| "completions/min_length": 85.0, |
| "epoch": 0.5128205128205128, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.923593731192653e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 60, |
| "step_time": 18.851675725076348 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 138.5, |
| "completions/min_length": 89.0, |
| "epoch": 0.5213675213675214, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 5.465298457922384, |
| "learning_rate": 9.919530398702916e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.625, |
| "reward_std": 0.46288391947746277, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 61, |
| "step_time": 23.250689563108608 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 179.0, |
| "completions/mean_length": 127.33333587646484, |
| "completions/min_length": 90.0, |
| "epoch": 0.5299145299145299, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8108509408876152, |
| "learning_rate": 9.915362677420044e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 62, |
| "step_time": 17.848797188140452 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 206.0, |
| "completions/mean_length": 120.29167175292969, |
| "completions/min_length": 76.0, |
| "epoch": 0.5384615384615384, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.308092195320936, |
| "learning_rate": 9.911090655779997e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 63, |
| "step_time": 17.623432660009712 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 248.0, |
| "completions/mean_length": 134.95834350585938, |
| "completions/min_length": 87.0, |
| "epoch": 0.5470085470085471, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5268436684785, |
| "learning_rate": 9.906714424431912e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 64, |
| "step_time": 17.328110087895766 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 136.0, |
| "completions/mean_length": 103.625, |
| "completions/min_length": 78.0, |
| "epoch": 0.5555555555555556, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.90223407623618e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 65, |
| "step_time": 16.78908740589395 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 411.0, |
| "completions/mean_length": 155.58334350585938, |
| "completions/min_length": 96.0, |
| "epoch": 0.5641025641025641, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.989847668888099, |
| "learning_rate": 9.897649706262473e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 66, |
| "step_time": 20.576793895103037 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 166.0, |
| "completions/mean_length": 116.375, |
| "completions/min_length": 72.0, |
| "epoch": 0.5726495726495726, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.892961411787723e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 67, |
| "step_time": 17.43586388300173 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 419.0, |
| "completions/mean_length": 143.9166717529297, |
| "completions/min_length": 77.0, |
| "epoch": 0.5811965811965812, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.888169292294075e-07, |
| "loss": 0.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 68, |
| "step_time": 19.43984568398446 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 162.83334350585938, |
| "completions/min_length": 82.0, |
| "epoch": 0.5897435897435898, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.434998918662544, |
| "learning_rate": 9.883273449466754e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 69, |
| "step_time": 23.73908931692131 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 388.0, |
| "completions/mean_length": 124.0, |
| "completions/min_length": 67.0, |
| "epoch": 0.5982905982905983, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.87827398719192e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 70, |
| "step_time": 20.64654406090267 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 265.0, |
| "completions/mean_length": 134.6666717529297, |
| "completions/min_length": 86.0, |
| "epoch": 0.6068376068376068, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5968105070039234, |
| "learning_rate": 9.87317101155446e-07, |
| "loss": 0.0, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 71, |
| "step_time": 20.010812991065904 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 308.0, |
| "completions/mean_length": 171.08334350585938, |
| "completions/min_length": 93.0, |
| "epoch": 0.6153846153846154, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.0926413930949215, |
| "learning_rate": 9.867964630835742e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.1666666716337204, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.1666666716337204, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 72, |
| "step_time": 20.434483774006367 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 266.0, |
| "completions/mean_length": 174.70834350585938, |
| "completions/min_length": 96.0, |
| "epoch": 0.6239316239316239, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.2991145791106025, |
| "learning_rate": 9.862654955511307e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.46288391947746277, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 73, |
| "step_time": 17.240905494894832 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 167.0, |
| "completions/mean_length": 111.45833587646484, |
| "completions/min_length": 78.0, |
| "epoch": 0.6324786324786325, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.857242098248542e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 74, |
| "step_time": 18.117783131077886 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 296.0, |
| "completions/mean_length": 167.75, |
| "completions/min_length": 94.0, |
| "epoch": 0.6410256410256411, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.244207236354717, |
| "learning_rate": 9.851726173904263e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 75, |
| "step_time": 15.151162466034293 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 202.0, |
| "completions/mean_length": 131.1666717529297, |
| "completions/min_length": 67.0, |
| "epoch": 0.6495726495726496, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.942120631967678, |
| "learning_rate": 9.846107299522303e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.5288647413253784, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 76, |
| "step_time": 18.624316211091354 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 179.0, |
| "completions/mean_length": 121.0, |
| "completions/min_length": 77.0, |
| "epoch": 0.6581196581196581, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.6815323593131066, |
| "learning_rate": 9.840385594331022e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 77, |
| "step_time": 18.551697693997994 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 169.0, |
| "completions/min_length": 77.0, |
| "epoch": 0.6666666666666666, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1096383939536474, |
| "learning_rate": 9.834561179740762e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 78, |
| "step_time": 24.29061187407933 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 377.0, |
| "completions/mean_length": 172.4166717529297, |
| "completions/min_length": 69.0, |
| "epoch": 0.6752136752136753, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.648200220206528, |
| "learning_rate": 9.82863417934129e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 79, |
| "step_time": 18.929019395960495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 297.0, |
| "completions/mean_length": 144.70834350585938, |
| "completions/min_length": 93.0, |
| "epoch": 0.6837606837606838, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.82260471889917e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 80, |
| "step_time": 20.19327363697812 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 192.0, |
| "completions/mean_length": 114.70833587646484, |
| "completions/min_length": 75.0, |
| "epoch": 0.6923076923076923, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.797725611704221, |
| "learning_rate": 9.816472926355086e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 81, |
| "step_time": 18.352771979989484 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 184.0, |
| "completions/mean_length": 123.33333587646484, |
| "completions/min_length": 81.0, |
| "epoch": 0.7008547008547008, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.601900556593981, |
| "learning_rate": 9.810238931821137e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 82, |
| "step_time": 17.430463393917307 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 236.0, |
| "completions/mean_length": 127.33333587646484, |
| "completions/min_length": 87.0, |
| "epoch": 0.7094017094017094, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 6.039152996604677, |
| "learning_rate": 9.803902867578072e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.45032864809036255, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 83, |
| "step_time": 18.738870390923694 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 204.0, |
| "completions/mean_length": 124.5, |
| "completions/min_length": 84.0, |
| "epoch": 0.717948717948718, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2511412523865073, |
| "learning_rate": 9.797464868072486e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 84, |
| "step_time": 14.673631524201483 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 199.0, |
| "completions/mean_length": 121.875, |
| "completions/min_length": 84.0, |
| "epoch": 0.7264957264957265, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5108332899361994, |
| "learning_rate": 9.79092506991396e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 85, |
| "step_time": 16.76843818486668 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 186.1666717529297, |
| "completions/min_length": 91.0, |
| "epoch": 0.7350427350427351, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1934024365374722, |
| "learning_rate": 9.784283611872168e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 86, |
| "step_time": 24.106740263057873 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 310.0, |
| "completions/mean_length": 133.75, |
| "completions/min_length": 80.0, |
| "epoch": 0.7435897435897436, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.946574408393729, |
| "learning_rate": 9.777540634873937e-07, |
| "loss": 0.0, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 87, |
| "step_time": 20.126178157981485 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 153.0, |
| "completions/mean_length": 114.125, |
| "completions/min_length": 77.0, |
| "epoch": 0.7521367521367521, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.770696282000244e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 88, |
| "step_time": 14.910376157145947 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 449.0, |
| "completions/mean_length": 159.45834350585938, |
| "completions/min_length": 89.0, |
| "epoch": 0.7606837606837606, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.447626606158567, |
| "learning_rate": 9.763750698483191e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.0416666679084301, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.0416666679084301, |
| "rewards/DirectReward/std": 0.20412413775920868, |
| "step": 89, |
| "step_time": 23.393353373976424 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 230.0, |
| "completions/mean_length": 137.4166717529297, |
| "completions/min_length": 91.0, |
| "epoch": 0.7692307692307693, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.756704031702919e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 90, |
| "step_time": 18.90628408291377 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 215.0, |
| "completions/mean_length": 129.1666717529297, |
| "completions/min_length": 70.0, |
| "epoch": 0.7777777777777778, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.721899610167876, |
| "learning_rate": 9.74955643118448e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 91, |
| "step_time": 18.222678838996217 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 207.0, |
| "completions/mean_length": 125.66667175292969, |
| "completions/min_length": 90.0, |
| "epoch": 0.7863247863247863, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.220458627927281, |
| "learning_rate": 9.742308048594664e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 92, |
| "step_time": 18.477360374992713 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 177.0, |
| "completions/mean_length": 111.875, |
| "completions/min_length": 79.0, |
| "epoch": 0.7948717948717948, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 4.535522820047623, |
| "learning_rate": 9.734959037738787e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 93, |
| "step_time": 17.09729927405715 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 205.0, |
| "completions/mean_length": 127.58333587646484, |
| "completions/min_length": 93.0, |
| "epoch": 0.8034188034188035, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.332086617112099, |
| "learning_rate": 9.727509554557415e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 94, |
| "step_time": 18.322427823906764 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 371.0, |
| "completions/mean_length": 138.08334350585938, |
| "completions/min_length": 74.0, |
| "epoch": 0.811965811965812, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.277983724049192, |
| "learning_rate": 9.719959757123071e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 95, |
| "step_time": 17.808966699056327 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 294.0, |
| "completions/mean_length": 170.375, |
| "completions/min_length": 116.0, |
| "epoch": 0.8205128205128205, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.739771167104894, |
| "learning_rate": 9.712309805636863e-07, |
| "loss": 0.0, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 96, |
| "step_time": 16.70554198510945 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 164.0, |
| "completions/min_length": 86.0, |
| "epoch": 0.8290598290598291, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.7045598624251e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 97, |
| "step_time": 22.844524731859565 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 160.45834350585938, |
| "completions/min_length": 93.0, |
| "epoch": 0.8376068376068376, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.7679202409348265, |
| "learning_rate": 9.69671009193584e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 98, |
| "step_time": 20.63621179596521 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 153.0, |
| "completions/mean_length": 108.54167175292969, |
| "completions/min_length": 72.0, |
| "epoch": 0.8461538461538461, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.688760660735402e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 99, |
| "step_time": 17.571437445934862 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 320.0, |
| "completions/mean_length": 132.625, |
| "completions/min_length": 72.0, |
| "epoch": 0.8547008547008547, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.680711737504829e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 100, |
| "step_time": 16.074826325057074 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 200.0, |
| "completions/mean_length": 133.08334350585938, |
| "completions/min_length": 100.0, |
| "epoch": 0.8632478632478633, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4623440161488626, |
| "learning_rate": 9.672563493036317e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.25, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.25, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 101, |
| "step_time": 18.10219144495204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 144.875, |
| "completions/min_length": 81.0, |
| "epoch": 0.8717948717948718, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.136388335123032, |
| "learning_rate": 9.664316100229576e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 102, |
| "step_time": 23.846886741928756 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 185.0, |
| "completions/mean_length": 133.1666717529297, |
| "completions/min_length": 100.0, |
| "epoch": 0.8803418803418803, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.553174452123047, |
| "learning_rate": 9.655969734088184e-07, |
| "loss": 7.450580596923828e-09, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 103, |
| "step_time": 18.368992744013667 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 171.2916717529297, |
| "completions/min_length": 104.0, |
| "epoch": 0.8888888888888888, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.030483780156879, |
| "learning_rate": 9.647524571715842e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 104, |
| "step_time": 20.49839890585281 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 157.0, |
| "completions/mean_length": 120.58333587646484, |
| "completions/min_length": 85.0, |
| "epoch": 0.8974358974358975, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.638980792312649e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 105, |
| "step_time": 17.887532850028947 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 241.20834350585938, |
| "completions/min_length": 137.0, |
| "epoch": 0.905982905982906, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1934178939465623, |
| "learning_rate": 9.63033857717128e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 106, |
| "step_time": 21.66952490177937 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 243.0, |
| "completions/mean_length": 119.66667175292969, |
| "completions/min_length": 72.0, |
| "epoch": 0.9145299145299145, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4029583947723014, |
| "learning_rate": 9.621598109673141e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 107, |
| "step_time": 18.755984880030155 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.20833333333333334, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 227.1666717529297, |
| "completions/min_length": 87.0, |
| "epoch": 0.9230769230769231, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.050633213976295, |
| "learning_rate": 9.612759575284481e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 108, |
| "step_time": 23.09442714881152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 199.125, |
| "completions/min_length": 91.0, |
| "epoch": 0.9316239316239316, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1856190703187184, |
| "learning_rate": 9.603823161552456e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.5, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 109, |
| "step_time": 22.7446844689548 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 300.0, |
| "completions/mean_length": 185.4166717529297, |
| "completions/min_length": 122.0, |
| "epoch": 0.9401709401709402, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.219253884511723, |
| "learning_rate": 9.594789058101153e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.45032867789268494, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 110, |
| "step_time": 19.432430485030636 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 300.0, |
| "completions/mean_length": 158.5, |
| "completions/min_length": 78.0, |
| "epoch": 0.9487179487179487, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.9434480698926624, |
| "learning_rate": 9.585657456627556e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 111, |
| "step_time": 19.8914508279413 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 230.0, |
| "completions/mean_length": 143.625, |
| "completions/min_length": 67.0, |
| "epoch": 0.9572649572649573, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.57642855089749e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 112, |
| "step_time": 18.41670674015768 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 294.0, |
| "completions/mean_length": 173.20834350585938, |
| "completions/min_length": 84.0, |
| "epoch": 0.9658119658119658, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.9730950809860361, |
| "learning_rate": 9.5671025367415e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 113, |
| "step_time": 19.967991027981043 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 257.0, |
| "completions/mean_length": 153.70834350585938, |
| "completions/min_length": 98.0, |
| "epoch": 0.9743589743589743, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4880675862830426, |
| "learning_rate": 9.557679612050707e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 114, |
| "step_time": 19.315066268201917 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 233.0, |
| "completions/mean_length": 148.0416717529297, |
| "completions/min_length": 99.0, |
| "epoch": 0.9829059829059829, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.727733266499768, |
| "learning_rate": 9.548159976772592e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.48112308979034424, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 115, |
| "step_time": 18.13044666009955 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 213.0, |
| "completions/mean_length": 144.4166717529297, |
| "completions/min_length": 89.0, |
| "epoch": 0.9914529914529915, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.673498845316457, |
| "learning_rate": 9.53854383290677e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 116, |
| "step_time": 18.670003678882495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 199.0, |
| "completions/mean_length": 122.45833587646484, |
| "completions/min_length": 92.0, |
| "epoch": 1.0, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.1204463796834565, |
| "learning_rate": 9.528831384500697e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 117, |
| "step_time": 18.861671580001712 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 181.0, |
| "completions/mean_length": 125.625, |
| "completions/min_length": 89.0, |
| "epoch": 1.0085470085470085, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.921297862681201, |
| "learning_rate": 9.519022837645336e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.1666666716337204, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.1666666716337204, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 118, |
| "step_time": 17.835945507977158 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 184.0, |
| "completions/mean_length": 122.58333587646484, |
| "completions/min_length": 78.0, |
| "epoch": 1.017094017094017, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7667070255524333, |
| "learning_rate": 9.509118400470792e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 119, |
| "step_time": 16.38893094402738 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 187.5, |
| "completions/min_length": 90.0, |
| "epoch": 1.0256410256410255, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.738805893529627, |
| "learning_rate": 9.499118283141886e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 120, |
| "step_time": 23.231575367972255 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 166.2916717529297, |
| "completions/min_length": 77.0, |
| "epoch": 1.0341880341880343, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 6.726665933761068, |
| "learning_rate": 9.489022697853708e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.5, |
| "reward_std": 0.48678088188171387, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 121, |
| "step_time": 21.582366964081302 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 244.0, |
| "completions/mean_length": 122.95833587646484, |
| "completions/min_length": 75.0, |
| "epoch": 1.0427350427350428, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4654365413609765, |
| "learning_rate": 9.478831858827103e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 122, |
| "step_time": 18.719794728094712 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 387.0, |
| "completions/mean_length": 156.58334350585938, |
| "completions/min_length": 98.0, |
| "epoch": 1.0512820512820513, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.518614968697279, |
| "learning_rate": 9.46854598230413e-07, |
| "loss": 0.0, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.4082186222076416, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 123, |
| "step_time": 16.624918844085187 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 171.58334350585938, |
| "completions/min_length": 81.0, |
| "epoch": 1.0598290598290598, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.8908124790031513, |
| "learning_rate": 9.458165286543476e-07, |
| "loss": 2.7318796114172983e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 124, |
| "step_time": 22.8522221269086 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 168.0, |
| "completions/mean_length": 116.66667175292969, |
| "completions/min_length": 82.0, |
| "epoch": 1.0683760683760684, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8711884639243928, |
| "learning_rate": 9.447689991815817e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 125, |
| "step_time": 17.05709482799284 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 284.0, |
| "completions/mean_length": 136.9166717529297, |
| "completions/min_length": 88.0, |
| "epoch": 1.0769230769230769, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8960742483414474, |
| "learning_rate": 9.437120320399157e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 126, |
| "step_time": 19.15497015300207 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 288.0, |
| "completions/mean_length": 141.9166717529297, |
| "completions/min_length": 78.0, |
| "epoch": 1.0854700854700854, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2167727746274357, |
| "learning_rate": 9.426456496574095e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 127, |
| "step_time": 16.232000201009214 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 239.0, |
| "completions/mean_length": 137.25, |
| "completions/min_length": 94.0, |
| "epoch": 1.0940170940170941, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.983643412750382, |
| "learning_rate": 9.415698746619079e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 128, |
| "step_time": 15.888537743827328 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 389.0, |
| "completions/mean_length": 141.5416717529297, |
| "completions/min_length": 86.0, |
| "epoch": 1.1025641025641026, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.165057749638586, |
| "learning_rate": 9.404847298805599e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 129, |
| "step_time": 18.227238620864227 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 221.70834350585938, |
| "completions/min_length": 88.0, |
| "epoch": 1.1111111111111112, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.393902383393346e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 130, |
| "step_time": 21.593748239101842 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 188.25, |
| "completions/min_length": 78.0, |
| "epoch": 1.1196581196581197, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.44404850872056, |
| "learning_rate": 9.38286423262532e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.375, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 131, |
| "step_time": 20.53462718287483 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 258.0, |
| "completions/mean_length": 163.6666717529297, |
| "completions/min_length": 99.0, |
| "epoch": 1.1282051282051282, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.142637478826627, |
| "learning_rate": 9.37173308072291e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.0833333358168602, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.0833333358168602, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 132, |
| "step_time": 17.820010513067245 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 288.0, |
| "completions/mean_length": 147.625, |
| "completions/min_length": 92.0, |
| "epoch": 1.1367521367521367, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.454648662265785, |
| "learning_rate": 9.360509163880919e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 133, |
| "step_time": 19.07073488808237 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 253.0, |
| "completions/mean_length": 154.375, |
| "completions/min_length": 103.0, |
| "epoch": 1.1452991452991452, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.167968743869224, |
| "learning_rate": 9.349192720262554e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.75, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 134, |
| "step_time": 18.33821720792912 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 281.0, |
| "completions/mean_length": 168.0416717529297, |
| "completions/min_length": 100.0, |
| "epoch": 1.1538461538461537, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.33778398999437e-07, |
| "loss": 0.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 135, |
| "step_time": 18.75077868811786 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 261.0, |
| "completions/mean_length": 138.875, |
| "completions/min_length": 85.0, |
| "epoch": 1.1623931623931625, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4909427051333943, |
| "learning_rate": 9.326283215161177e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 136, |
| "step_time": 20.035130331059918 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 155.9166717529297, |
| "completions/min_length": 95.0, |
| "epoch": 1.170940170940171, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5890848024115303, |
| "learning_rate": 9.314690639800905e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 137, |
| "step_time": 22.158014100044966 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 285.0, |
| "completions/mean_length": 146.9166717529297, |
| "completions/min_length": 96.0, |
| "epoch": 1.1794871794871795, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.303006509899418e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 138, |
| "step_time": 18.06703043100424 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 509.0, |
| "completions/mean_length": 170.6666717529297, |
| "completions/min_length": 98.0, |
| "epoch": 1.188034188034188, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.441856720808841, |
| "learning_rate": 9.291231073385306e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.75, |
| "reward_std": 0.41387641429901123, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 139, |
| "step_time": 22.797705161152408 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 238.83334350585938, |
| "completions/min_length": 113.0, |
| "epoch": 1.1965811965811965, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.156395154261485, |
| "learning_rate": 9.279364580124614e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 140, |
| "step_time": 23.603184598963708 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 206.0, |
| "completions/min_length": 109.0, |
| "epoch": 1.205128205128205, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1006876370219505, |
| "learning_rate": 9.267407281915541e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 141, |
| "step_time": 19.2320515147876 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 497.0, |
| "completions/mean_length": 170.83334350585938, |
| "completions/min_length": 90.0, |
| "epoch": 1.2136752136752136, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.712630337594954, |
| "learning_rate": 9.255359432483105e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.625, |
| "reward_std": 0.46288391947746277, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 142, |
| "step_time": 22.708583839936182 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 161.5416717529297, |
| "completions/min_length": 85.0, |
| "epoch": 1.2222222222222223, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.231155116073906, |
| "learning_rate": 9.243221287473755e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 143, |
| "step_time": 21.67199845891446 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 196.125, |
| "completions/min_length": 92.0, |
| "epoch": 1.2307692307692308, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.804153280133587, |
| "learning_rate": 9.230993104449938e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 144, |
| "step_time": 22.028480096021667 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 217.5, |
| "completions/min_length": 111.0, |
| "epoch": 1.2393162393162394, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.788338753188895, |
| "learning_rate": 9.218675142884646e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.5, |
| "reward_std": 0.4446708858013153, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 145, |
| "step_time": 19.534104684134945 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 292.0, |
| "completions/mean_length": 153.58334350585938, |
| "completions/min_length": 93.0, |
| "epoch": 1.2478632478632479, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8357807648003805, |
| "learning_rate": 9.206267664155906e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 146, |
| "step_time": 19.197380932047963 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 367.0, |
| "completions/mean_length": 244.75, |
| "completions/min_length": 126.0, |
| "epoch": 1.2564102564102564, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1727582774739687, |
| "learning_rate": 9.193770931541229e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 147, |
| "step_time": 17.627865070011467 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 237.4166717529297, |
| "completions/min_length": 118.0, |
| "epoch": 1.264957264957265, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.187570863709847, |
| "learning_rate": 9.181185210212034e-07, |
| "loss": 0.0, |
| "reward": 0.25, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.25, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 148, |
| "step_time": 20.21642567892559 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 333.0, |
| "completions/mean_length": 149.6666717529297, |
| "completions/min_length": 113.0, |
| "epoch": 1.2735042735042734, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.168510767228006e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 149, |
| "step_time": 18.976538405986503 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 484.0, |
| "completions/mean_length": 166.95834350585938, |
| "completions/min_length": 86.0, |
| "epoch": 1.282051282051282, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.305924182613328, |
| "learning_rate": 9.155747871531443e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 150, |
| "step_time": 18.358726338017732 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 280.0, |
| "completions/mean_length": 162.83334350585938, |
| "completions/min_length": 99.0, |
| "epoch": 1.2905982905982907, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5238418938493026, |
| "learning_rate": 9.142896793941546e-07, |
| "loss": 0.0, |
| "reward": 0.0833333358168602, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.0833333358168602, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 151, |
| "step_time": 16.1946898100432 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/mean_length": 135.875, |
| "completions/min_length": 86.0, |
| "epoch": 1.2991452991452992, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4272994999255904, |
| "learning_rate": 9.129957807148665e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 152, |
| "step_time": 17.4437215058133 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 220.20834350585938, |
| "completions/min_length": 102.0, |
| "epoch": 1.3076923076923077, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.829670536018602, |
| "learning_rate": 9.116931185708523e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.375, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 153, |
| "step_time": 22.512292249826714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 228.0, |
| "completions/mean_length": 150.33334350585938, |
| "completions/min_length": 103.0, |
| "epoch": 1.3162393162393162, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.34912343372272, |
| "learning_rate": 9.103817206036382e-07, |
| "loss": 0.0, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 154, |
| "step_time": 18.244432791834697 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 434.0, |
| "completions/mean_length": 185.58334350585938, |
| "completions/min_length": 79.0, |
| "epoch": 1.3247863247863247, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5840875277457727, |
| "learning_rate": 9.090616146401183e-07, |
| "loss": 0.0, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 155, |
| "step_time": 21.505654674954712 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 345.0, |
| "completions/mean_length": 201.4166717529297, |
| "completions/min_length": 102.0, |
| "epoch": 1.3333333333333333, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.077328286919636e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 156, |
| "step_time": 19.37729303818196 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 268.125, |
| "completions/min_length": 99.0, |
| "epoch": 1.341880341880342, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.440245697597662, |
| "learning_rate": 9.063953909550288e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.625, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 157, |
| "step_time": 23.063575848005712 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 387.0, |
| "completions/mean_length": 158.6666717529297, |
| "completions/min_length": 92.0, |
| "epoch": 1.3504273504273505, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.9286911707727092, |
| "learning_rate": 9.050493298087522e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 158, |
| "step_time": 18.669896906940266 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 207.2916717529297, |
| "completions/min_length": 121.0, |
| "epoch": 1.358974358974359, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.8168127884790817, |
| "learning_rate": 9.036946738155547e-07, |
| "loss": 0.0, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 159, |
| "step_time": 20.597389525035396 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 293.0, |
| "completions/mean_length": 157.6666717529297, |
| "completions/min_length": 110.0, |
| "epoch": 1.3675213675213675, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 9.02331451720234e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 160, |
| "step_time": 18.54589762305841 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 227.5, |
| "completions/min_length": 95.0, |
| "epoch": 1.376068376068376, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.59556289577876, |
| "learning_rate": 9.009596924493535e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 161, |
| "step_time": 21.296804127050564 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 190.7916717529297, |
| "completions/min_length": 102.0, |
| "epoch": 1.3846153846153846, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.0464267285160695, |
| "learning_rate": 8.995794251106294e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 162, |
| "step_time": 23.35893664485775 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 387.0, |
| "completions/mean_length": 169.0416717529297, |
| "completions/min_length": 98.0, |
| "epoch": 1.393162393162393, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.98190678992313e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 163, |
| "step_time": 17.475462882081047 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 392.0, |
| "completions/mean_length": 208.95834350585938, |
| "completions/min_length": 125.0, |
| "epoch": 1.4017094017094016, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.7616503264346504, |
| "learning_rate": 8.967934835625688e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 164, |
| "step_time": 19.14912751899101 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 228.0, |
| "completions/mean_length": 158.1666717529297, |
| "completions/min_length": 83.0, |
| "epoch": 1.4102564102564101, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.062994830109228, |
| "learning_rate": 8.953878684688492e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 165, |
| "step_time": 17.250508294906467 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 443.0, |
| "completions/mean_length": 202.5, |
| "completions/min_length": 95.0, |
| "epoch": 1.4188034188034189, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.0014953090631384, |
| "learning_rate": 8.939738635372663e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 166, |
| "step_time": 20.918956113047898 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 197.0, |
| "completions/min_length": 119.0, |
| "epoch": 1.4273504273504274, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.6251124890212356, |
| "learning_rate": 8.925514987719578e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 167, |
| "step_time": 20.796506118029356 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 223.0, |
| "completions/mean_length": 141.5, |
| "completions/min_length": 98.0, |
| "epoch": 1.435897435897436, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7166701443959904, |
| "learning_rate": 8.911208043544511e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 168, |
| "step_time": 18.55257878289558 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 282.0, |
| "completions/mean_length": 194.5, |
| "completions/min_length": 112.0, |
| "epoch": 1.4444444444444444, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.896818106430224e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 169, |
| "step_time": 18.129941303050146 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 394.0, |
| "completions/mean_length": 155.875, |
| "completions/min_length": 100.0, |
| "epoch": 1.452991452991453, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8931154598320012, |
| "learning_rate": 8.882345481720532e-07, |
| "loss": 0.0, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 170, |
| "step_time": 17.792866962961853 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 498.0, |
| "completions/mean_length": 202.5, |
| "completions/min_length": 104.0, |
| "epoch": 1.4615384615384617, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8077203478139163, |
| "learning_rate": 8.867790476513817e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 171, |
| "step_time": 22.304877602960914 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 191.0, |
| "completions/mean_length": 140.83334350585938, |
| "completions/min_length": 113.0, |
| "epoch": 1.4700854700854702, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.853153399656512e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 172, |
| "step_time": 17.420097013004124 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 245.0, |
| "completions/mean_length": 165.33334350585938, |
| "completions/min_length": 118.0, |
| "epoch": 1.4786324786324787, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.549540981178653, |
| "learning_rate": 8.838434561736555e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.34503278136253357, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 173, |
| "step_time": 17.987250686157495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 223.0416717529297, |
| "completions/min_length": 122.0, |
| "epoch": 1.4871794871794872, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.4343509416204703, |
| "learning_rate": 8.823634275076791e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 174, |
| "step_time": 18.24587313295342 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 268.7083435058594, |
| "completions/min_length": 144.0, |
| "epoch": 1.4957264957264957, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.252545836246004, |
| "learning_rate": 8.808752853728341e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.625, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 175, |
| "step_time": 22.467531039146706 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 313.0, |
| "completions/mean_length": 184.7916717529297, |
| "completions/min_length": 112.0, |
| "epoch": 1.5042735042735043, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.7344161780479455, |
| "learning_rate": 8.793790613463954e-07, |
| "loss": 0.0, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 176, |
| "step_time": 19.76033266214654 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 270.0, |
| "completions/mean_length": 177.875, |
| "completions/min_length": 112.0, |
| "epoch": 1.5128205128205128, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.181744975199654, |
| "learning_rate": 8.778747871771291e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.1666666716337204, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.1666666716337204, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 177, |
| "step_time": 18.015875509940088 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 172.875, |
| "completions/min_length": 110.0, |
| "epoch": 1.5213675213675213, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.4706045996200867, |
| "learning_rate": 8.763624947846194e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 178, |
| "step_time": 19.885289455065504 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 209.0416717529297, |
| "completions/min_length": 107.0, |
| "epoch": 1.5299145299145298, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.179320530042256, |
| "learning_rate": 8.748422162585913e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.5, |
| "reward_std": 0.5232069492340088, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 179, |
| "step_time": 22.732836863957345 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 271.5, |
| "completions/min_length": 125.0, |
| "epoch": 1.5384615384615383, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.733139838582298e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 180, |
| "step_time": 18.790694013005123 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 340.0, |
| "completions/mean_length": 223.1666717529297, |
| "completions/min_length": 126.0, |
| "epoch": 1.547008547008547, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.497109256892215, |
| "learning_rate": 8.717778300114951e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.375, |
| "reward_std": 0.48112308979034424, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 181, |
| "step_time": 19.298166180960834 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 431.0, |
| "completions/mean_length": 181.125, |
| "completions/min_length": 124.0, |
| "epoch": 1.5555555555555556, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.143088361021751, |
| "learning_rate": 8.702337873144342e-07, |
| "loss": 0.0, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 182, |
| "step_time": 18.57162966602482 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 375.0, |
| "completions/mean_length": 247.33334350585938, |
| "completions/min_length": 119.0, |
| "epoch": 1.564102564102564, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.714219085533317, |
| "learning_rate": 8.686818885304905e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 183, |
| "step_time": 20.204337507020682 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 374.0, |
| "completions/mean_length": 175.375, |
| "completions/min_length": 93.0, |
| "epoch": 1.5726495726495726, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.671221665898073e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 184, |
| "step_time": 20.015477567911148 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 285.0, |
| "completions/mean_length": 186.20834350585938, |
| "completions/min_length": 111.0, |
| "epoch": 1.5811965811965814, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.407506218204185, |
| "learning_rate": 8.655546545885293e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 185, |
| "step_time": 18.046158305136487 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 326.0, |
| "completions/mean_length": 196.08334350585938, |
| "completions/min_length": 107.0, |
| "epoch": 1.5897435897435899, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.058516359036902, |
| "learning_rate": 8.63979385788101e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 186, |
| "step_time": 18.463308996055275 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 364.0, |
| "completions/mean_length": 184.58334350585938, |
| "completions/min_length": 138.0, |
| "epoch": 1.5982905982905984, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.846450377550318, |
| "learning_rate": 8.623963936145599e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 187, |
| "step_time": 19.909831075929105 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 208.25, |
| "completions/min_length": 117.0, |
| "epoch": 1.606837606837607, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5664137760462027, |
| "learning_rate": 8.608057116578282e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 188, |
| "step_time": 19.99824434798211 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 244.08334350585938, |
| "completions/min_length": 114.0, |
| "epoch": 1.6153846153846154, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.5878253651845797, |
| "learning_rate": 8.592073736709995e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 189, |
| "step_time": 22.509123417083174 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 204.4166717529297, |
| "completions/min_length": 116.0, |
| "epoch": 1.623931623931624, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.3926050936269085, |
| "learning_rate": 8.576014135696226e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 190, |
| "step_time": 23.23409419809468 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 320.0, |
| "completions/mean_length": 190.75, |
| "completions/min_length": 117.0, |
| "epoch": 1.6324786324786325, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8645845842973827, |
| "learning_rate": 8.559878654309818e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 191, |
| "step_time": 17.18606996885501 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 335.0, |
| "completions/mean_length": 192.0, |
| "completions/min_length": 112.0, |
| "epoch": 1.641025641025641, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.944740178751773, |
| "learning_rate": 8.543667634933741e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 192, |
| "step_time": 15.3680785309989 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 246.7916717529297, |
| "completions/min_length": 121.0, |
| "epoch": 1.6495726495726495, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.2852388234161536, |
| "learning_rate": 8.527381421553828e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 193, |
| "step_time": 22.873547612922266 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 401.0, |
| "completions/mean_length": 238.375, |
| "completions/min_length": 134.0, |
| "epoch": 1.658119658119658, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.307557598262501, |
| "learning_rate": 8.511020359751466e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 194, |
| "step_time": 20.949345903936774 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 210.625, |
| "completions/min_length": 96.0, |
| "epoch": 1.6666666666666665, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.9063227311383588, |
| "learning_rate": 8.49458479669628e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 195, |
| "step_time": 22.927933229133487 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 322.0, |
| "completions/mean_length": 196.83334350585938, |
| "completions/min_length": 109.0, |
| "epoch": 1.6752136752136753, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.136272435696416, |
| "learning_rate": 8.478075081138745e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.625, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 196, |
| "step_time": 20.09887309302576 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 196.0, |
| "completions/mean_length": 144.875, |
| "completions/min_length": 97.0, |
| "epoch": 1.6837606837606838, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7261698455521928, |
| "learning_rate": 8.461491563402806e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.35634833574295044, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 197, |
| "step_time": 15.538168675964698 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 401.0, |
| "completions/mean_length": 195.5416717529297, |
| "completions/min_length": 107.0, |
| "epoch": 1.6923076923076923, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.549433123227539, |
| "learning_rate": 8.444834595378433e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 198, |
| "step_time": 21.125746320933104 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 204.1666717529297, |
| "completions/min_length": 115.0, |
| "epoch": 1.7008547008547008, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.160619823441638, |
| "learning_rate": 8.428104530514154e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 199, |
| "step_time": 22.330627373885363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 214.25, |
| "completions/min_length": 95.0, |
| "epoch": 1.7094017094017095, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.7017741620270974, |
| "learning_rate": 8.411301723809563e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 200, |
| "step_time": 22.8249277099967 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 415.0, |
| "completions/mean_length": 187.45834350585938, |
| "completions/min_length": 119.0, |
| "epoch": 1.717948717948718, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.341028197900142, |
| "learning_rate": 8.394426531807777e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 201, |
| "step_time": 23.711466323118657 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 253.9166717529297, |
| "completions/min_length": 126.0, |
| "epoch": 1.7264957264957266, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.5947202215133074, |
| "learning_rate": 8.377479312587879e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 202, |
| "step_time": 23.298730823909864 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 265.75, |
| "completions/min_length": 159.0, |
| "epoch": 1.735042735042735, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.434580473115701, |
| "learning_rate": 8.360460425757314e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.42645785212516785, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 203, |
| "step_time": 24.612847085809335 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 347.0, |
| "completions/mean_length": 208.70834350585938, |
| "completions/min_length": 135.0, |
| "epoch": 1.7435897435897436, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.3349678640080507, |
| "learning_rate": 8.34337023244426e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 204, |
| "step_time": 20.873330239206553 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 292.16668701171875, |
| "completions/min_length": 142.0, |
| "epoch": 1.7521367521367521, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.6307724081524055, |
| "learning_rate": 8.326209095289971e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.1666666716337204, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.1666666716337204, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 205, |
| "step_time": 20.693512301892042 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 325.0, |
| "completions/mean_length": 174.9166717529297, |
| "completions/min_length": 91.0, |
| "epoch": 1.7606837606837606, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.411344030298005, |
| "learning_rate": 8.308977378441071e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 206, |
| "step_time": 20.758061395026743 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 344.0, |
| "completions/mean_length": 159.875, |
| "completions/min_length": 106.0, |
| "epoch": 1.7692307692307692, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.232624378646607, |
| "learning_rate": 8.291675447541833e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 207, |
| "step_time": 21.185840863967314 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 312.0, |
| "completions/mean_length": 208.4166717529297, |
| "completions/min_length": 122.0, |
| "epoch": 1.7777777777777777, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.274303669726426e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 208, |
| "step_time": 17.19423439214006 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 215.875, |
| "completions/min_length": 108.0, |
| "epoch": 1.7863247863247862, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.4907102533613914, |
| "learning_rate": 8.256862413611112e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 209, |
| "step_time": 22.47058302303776 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 237.6666717529297, |
| "completions/min_length": 136.0, |
| "epoch": 1.7948717948717947, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.228696973112516, |
| "learning_rate": 8.239352049286435e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 210, |
| "step_time": 24.07968674483709 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 199.25, |
| "completions/min_length": 132.0, |
| "epoch": 1.8034188034188035, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.051670949641414, |
| "learning_rate": 8.221772948309361e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 211, |
| "step_time": 22.5069483818952 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 286.5, |
| "completions/min_length": 131.0, |
| "epoch": 1.811965811965812, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.364405811417097, |
| "learning_rate": 8.204125483695403e-07, |
| "loss": -2.2351741790771484e-08, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.5232069492340088, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 212, |
| "step_time": 19.9161701539997 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 258.875, |
| "completions/min_length": 125.0, |
| "epoch": 1.8205128205128205, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 2.910315957098234, |
| "learning_rate": 8.186410029910693e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.39000558853149414, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 213, |
| "step_time": 24.175829170038924 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 474.0, |
| "completions/mean_length": 219.70834350585938, |
| "completions/min_length": 106.0, |
| "epoch": 1.8290598290598292, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1343631146415865, |
| "learning_rate": 8.168626962864045e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 214, |
| "step_time": 22.442220974015072 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 252.0, |
| "completions/mean_length": 183.125, |
| "completions/min_length": 107.0, |
| "epoch": 1.8376068376068377, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.018534731458141, |
| "learning_rate": 8.150776659898979e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 215, |
| "step_time": 19.490405602846295 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 268.3333435058594, |
| "completions/min_length": 126.0, |
| "epoch": 1.8461538461538463, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.4897278456757723, |
| "learning_rate": 8.132859499785707e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 216, |
| "step_time": 23.004686592146754 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 458.0, |
| "completions/mean_length": 262.4583435058594, |
| "completions/min_length": 104.0, |
| "epoch": 1.8547008547008548, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.519622253062086, |
| "learning_rate": 8.114875862713105e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.625, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 217, |
| "step_time": 21.95731167285703 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 269.625, |
| "completions/min_length": 123.0, |
| "epoch": 1.8632478632478633, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8296703489390518, |
| "learning_rate": 8.096826130280639e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 218, |
| "step_time": 20.837155556073412 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 211.125, |
| "completions/min_length": 129.0, |
| "epoch": 1.8717948717948718, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.00819362311963, |
| "learning_rate": 8.078710685490264e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 219, |
| "step_time": 22.848018951015547 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 222.0416717529297, |
| "completions/min_length": 116.0, |
| "epoch": 1.8803418803418803, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.6314990122518758, |
| "learning_rate": 8.060529912738314e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 220, |
| "step_time": 23.736944764852524 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 484.0, |
| "completions/mean_length": 197.08334350585938, |
| "completions/min_length": 131.0, |
| "epoch": 1.8888888888888888, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.11241625500478, |
| "learning_rate": 8.042284197807323e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 221, |
| "step_time": 19.788710489869118 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.2916666666666667, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 277.16668701171875, |
| "completions/min_length": 95.0, |
| "epoch": 1.8974358974358974, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.813314175195223, |
| "learning_rate": 8.023973927857857e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 222, |
| "step_time": 20.376326735829934 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 354.0, |
| "completions/mean_length": 191.125, |
| "completions/min_length": 124.0, |
| "epoch": 1.9059829059829059, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 8.005599491420288e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 223, |
| "step_time": 20.379577895160764 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 237.7916717529297, |
| "completions/min_length": 140.0, |
| "epoch": 1.9145299145299144, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.152257863184043, |
| "learning_rate": 7.987161278386554e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 224, |
| "step_time": 24.762638071086258 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 252.75, |
| "completions/min_length": 154.0, |
| "epoch": 1.9230769230769231, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8619015052982169, |
| "learning_rate": 7.968659680001886e-07, |
| "loss": 0.0, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 225, |
| "step_time": 20.655045981984586 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 395.0, |
| "completions/mean_length": 233.7916717529297, |
| "completions/min_length": 147.0, |
| "epoch": 1.9316239316239316, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1416574705294376, |
| "learning_rate": 7.950095088856508e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 226, |
| "step_time": 19.113854472059757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 204.4166717529297, |
| "completions/min_length": 136.0, |
| "epoch": 1.9401709401709402, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1405440056659866, |
| "learning_rate": 7.931467898877297e-07, |
| "loss": 2.9802322387695312e-08, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 227, |
| "step_time": 22.203701745951548 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 228.0, |
| "completions/mean_length": 184.1666717529297, |
| "completions/min_length": 134.0, |
| "epoch": 1.9487179487179487, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.4102170477270977, |
| "learning_rate": 7.912778505319435e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 228, |
| "step_time": 19.265432573854923 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 277.04168701171875, |
| "completions/min_length": 171.0, |
| "epoch": 1.9572649572649574, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8225846532460497, |
| "learning_rate": 7.894027304758021e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 229, |
| "step_time": 19.273440279997885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 251.20834350585938, |
| "completions/min_length": 155.0, |
| "epoch": 1.965811965811966, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.7533125679792994, |
| "learning_rate": 7.875214695079646e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 230, |
| "step_time": 23.564144550124183 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 248.375, |
| "completions/min_length": 132.0, |
| "epoch": 1.9743589743589745, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.5622448185452433, |
| "learning_rate": 7.856341075473961e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.45032864809036255, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 231, |
| "step_time": 20.24388805613853 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 190.875, |
| "completions/min_length": 117.0, |
| "epoch": 1.982905982905983, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.837406846425203e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 232, |
| "step_time": 20.42177312099375 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 227.875, |
| "completions/min_length": 138.0, |
| "epoch": 1.9914529914529915, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.8326947708193893, |
| "learning_rate": 7.818412409703694e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.25, |
| "reward_std": 0.34503278136253357, |
| "rewards/DirectReward/mean": 0.25, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 233, |
| "step_time": 20.782191360834986 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.20833333333333334, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 268.875, |
| "completions/min_length": 135.0, |
| "epoch": 2.0, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.0196629381102635, |
| "learning_rate": 7.799358168357322e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 234, |
| "step_time": 21.320435292087495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 256.66668701171875, |
| "completions/min_length": 138.0, |
| "epoch": 2.0085470085470085, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.2108604536388268, |
| "learning_rate": 7.780244526702979e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.5, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 235, |
| "step_time": 23.034640240017325 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 240.0, |
| "completions/mean_length": 154.9166717529297, |
| "completions/min_length": 115.0, |
| "epoch": 2.017094017094017, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.761071890317994e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 236, |
| "step_time": 17.877617254154757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 384.0, |
| "completions/mean_length": 211.33334350585938, |
| "completions/min_length": 107.0, |
| "epoch": 2.0256410256410255, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.741840666031516e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 237, |
| "step_time": 21.11333019915037 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 285.0, |
| "completions/mean_length": 178.5416717529297, |
| "completions/min_length": 131.0, |
| "epoch": 2.034188034188034, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4310328479056436, |
| "learning_rate": 7.72255126191589e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 238, |
| "step_time": 18.76558225415647 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.20833333333333334, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 269.29168701171875, |
| "completions/min_length": 147.0, |
| "epoch": 2.0427350427350426, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.2598807868943944, |
| "learning_rate": 7.703204087277988e-07, |
| "loss": 2.2351741790771484e-08, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 239, |
| "step_time": 19.557289382908493 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 479.0, |
| "completions/mean_length": 217.1666717529297, |
| "completions/min_length": 108.0, |
| "epoch": 2.051282051282051, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.008189621537427, |
| "learning_rate": 7.683799552650534e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 240, |
| "step_time": 21.52867266908288 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 215.875, |
| "completions/min_length": 121.0, |
| "epoch": 2.0598290598290596, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.664338069783389e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 241, |
| "step_time": 22.614858294138685 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 274.0, |
| "completions/mean_length": 187.75, |
| "completions/min_length": 110.0, |
| "epoch": 2.0683760683760686, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.917472209335309, |
| "learning_rate": 7.644820051634812e-07, |
| "loss": 0.0, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 242, |
| "step_time": 19.406687207054347 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 243.25, |
| "completions/min_length": 153.0, |
| "epoch": 2.076923076923077, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.091667915907802, |
| "learning_rate": 7.625245912362697e-07, |
| "loss": 0.0, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 243, |
| "step_time": 22.722663837019354 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 374.0, |
| "completions/mean_length": 196.7916717529297, |
| "completions/min_length": 136.0, |
| "epoch": 2.0854700854700856, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.605616067315792e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 244, |
| "step_time": 20.115028999047354 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 237.2916717529297, |
| "completions/min_length": 160.0, |
| "epoch": 2.094017094017094, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.534475300068824, |
| "learning_rate": 7.585930933024873e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.3535533845424652, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 245, |
| "step_time": 22.537130105076358 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 302.91668701171875, |
| "completions/min_length": 160.0, |
| "epoch": 2.1025641025641026, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.585044693136781, |
| "learning_rate": 7.56619092719392e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 246, |
| "step_time": 20.0160381749738 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 418.0, |
| "completions/mean_length": 213.625, |
| "completions/min_length": 153.0, |
| "epoch": 2.111111111111111, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.546396468691241e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 247, |
| "step_time": 18.54481313517317 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 273.79168701171875, |
| "completions/min_length": 136.0, |
| "epoch": 2.1196581196581197, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.9842633019108633, |
| "learning_rate": 7.526547977540592e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.5, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 248, |
| "step_time": 22.993283902993426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 411.0, |
| "completions/mean_length": 212.5416717529297, |
| "completions/min_length": 151.0, |
| "epoch": 2.128205128205128, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.252521271842397, |
| "learning_rate": 7.506645874912263e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 249, |
| "step_time": 20.962747680023313 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 482.0, |
| "completions/mean_length": 214.375, |
| "completions/min_length": 134.0, |
| "epoch": 2.1367521367521367, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.486690583114136e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 250, |
| "step_time": 19.538640187121928 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 285.0, |
| "completions/mean_length": 194.625, |
| "completions/min_length": 156.0, |
| "epoch": 2.1452991452991452, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.466682525582731e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 251, |
| "step_time": 21.22228294890374 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 215.7916717529297, |
| "completions/min_length": 145.0, |
| "epoch": 2.1538461538461537, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8518710267030916, |
| "learning_rate": 7.446622126874218e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 252, |
| "step_time": 22.823727544164285 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 256.375, |
| "completions/min_length": 156.0, |
| "epoch": 2.1623931623931623, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.532200696634518, |
| "learning_rate": 7.426509812655405e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 253, |
| "step_time": 22.93512307992205 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 492.0, |
| "completions/mean_length": 249.0, |
| "completions/min_length": 164.0, |
| "epoch": 2.1709401709401708, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.8724148192552543, |
| "learning_rate": 7.406346009694711e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 254, |
| "step_time": 22.645130150951445 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 480.0, |
| "completions/mean_length": 226.375, |
| "completions/min_length": 147.0, |
| "epoch": 2.1794871794871793, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.632751425398144, |
| "learning_rate": 7.38613114585311e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5, |
| "reward_std": 0.41387641429901123, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 255, |
| "step_time": 22.063507050042972 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 294.0, |
| "completions/mean_length": 182.83334350585938, |
| "completions/min_length": 121.0, |
| "epoch": 2.1880341880341883, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.992299246702961, |
| "learning_rate": 7.365865650075045e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 256, |
| "step_time": 19.747207909123972 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 246.0, |
| "completions/mean_length": 169.83334350585938, |
| "completions/min_length": 124.0, |
| "epoch": 2.1965811965811968, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.864408840244053, |
| "learning_rate": 7.345549952379333e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 257, |
| "step_time": 15.296442082850263 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 243.0, |
| "completions/mean_length": 178.75, |
| "completions/min_length": 144.0, |
| "epoch": 2.2051282051282053, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.325184483850042e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 258, |
| "step_time": 18.195927061140537 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 502.0, |
| "completions/mean_length": 228.5, |
| "completions/min_length": 122.0, |
| "epoch": 2.213675213675214, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.172288661025458, |
| "learning_rate": 7.304769676627338e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.48112308979034424, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 259, |
| "step_time": 21.792258307803422 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 466.0, |
| "completions/mean_length": 198.625, |
| "completions/min_length": 110.0, |
| "epoch": 2.2222222222222223, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 7.284305963898314e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 260, |
| "step_time": 22.21475754189305 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 300.41668701171875, |
| "completions/min_length": 163.0, |
| "epoch": 2.230769230769231, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.283611926052555, |
| "learning_rate": 7.263793779887808e-07, |
| "loss": -2.7318796114172983e-08, |
| "reward": 0.375, |
| "reward_std": 0.48112308979034424, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 261, |
| "step_time": 20.217762012034655 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 262.41668701171875, |
| "completions/min_length": 135.0, |
| "epoch": 2.2393162393162394, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.3821252988360473, |
| "learning_rate": 7.243233559849177e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 262, |
| "step_time": 21.31954277609475 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 360.0, |
| "completions/mean_length": 220.25, |
| "completions/min_length": 151.0, |
| "epoch": 2.247863247863248, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.8617464546031854, |
| "learning_rate": 7.222625740055071e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 263, |
| "step_time": 19.23293857020326 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 218.20834350585938, |
| "completions/min_length": 149.0, |
| "epoch": 2.2564102564102564, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7179133974670378, |
| "learning_rate": 7.201970757788171e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 264, |
| "step_time": 19.067095732083544 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 290.0, |
| "completions/mean_length": 180.5416717529297, |
| "completions/min_length": 102.0, |
| "epoch": 2.264957264957265, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7404935397079053, |
| "learning_rate": 7.181269051331909e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 265, |
| "step_time": 17.151774026919156 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 498.0, |
| "completions/mean_length": 198.83334350585938, |
| "completions/min_length": 123.0, |
| "epoch": 2.2735042735042734, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.696380335485314, |
| "learning_rate": 7.160521059961168e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 266, |
| "step_time": 22.052734824828804 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 300.0, |
| "completions/mean_length": 205.25, |
| "completions/min_length": 153.0, |
| "epoch": 2.282051282051282, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.9653237290603682, |
| "learning_rate": 7.139727223932968e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.5, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 267, |
| "step_time": 19.67189479805529 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 199.08334350585938, |
| "completions/min_length": 102.0, |
| "epoch": 2.2905982905982905, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1222496549721424, |
| "learning_rate": 7.118887984477116e-07, |
| "loss": -2.7318796114172983e-08, |
| "reward": 0.5, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 268, |
| "step_time": 21.68394836387597 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 253.0416717529297, |
| "completions/min_length": 123.0, |
| "epoch": 2.299145299145299, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.0385045786827827, |
| "learning_rate": 7.098003783786844e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 269, |
| "step_time": 19.445077328011394 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 242.33334350585938, |
| "completions/min_length": 148.0, |
| "epoch": 2.3076923076923075, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.7195103261370186, |
| "learning_rate": 7.077075065009433e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 270, |
| "step_time": 23.04076889483258 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 257.0, |
| "completions/mean_length": 161.70834350585938, |
| "completions/min_length": 106.0, |
| "epoch": 2.316239316239316, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5492475762479145, |
| "learning_rate": 7.056102272236798e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 271, |
| "step_time": 16.340166823007166 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 250.0, |
| "completions/mean_length": 160.0, |
| "completions/min_length": 97.0, |
| "epoch": 2.324786324786325, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.016277713702069, |
| "learning_rate": 7.035085850496079e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 272, |
| "step_time": 20.403489847900346 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 403.0, |
| "completions/mean_length": 206.5416717529297, |
| "completions/min_length": 119.0, |
| "epoch": 2.3333333333333335, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.3076076518289272, |
| "learning_rate": 7.014026245740184e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 273, |
| "step_time": 20.88275165692903 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 438.0, |
| "completions/mean_length": 183.25, |
| "completions/min_length": 129.0, |
| "epoch": 2.341880341880342, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.9746440909378573, |
| "learning_rate": 6.99292390483834e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 274, |
| "step_time": 18.750420602038503 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 483.0, |
| "completions/mean_length": 196.70834350585938, |
| "completions/min_length": 109.0, |
| "epoch": 2.3504273504273505, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.971779275566593e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 275, |
| "step_time": 21.708959653042257 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 215.0, |
| "completions/mean_length": 152.70834350585938, |
| "completions/min_length": 115.0, |
| "epoch": 2.358974358974359, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.950592806598326e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 276, |
| "step_time": 17.488857438089326 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 210.4166717529297, |
| "completions/min_length": 122.0, |
| "epoch": 2.3675213675213675, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.3572443176798923, |
| "learning_rate": 6.929364947494727e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.375, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 277, |
| "step_time": 23.755606928141788 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 276.0, |
| "completions/mean_length": 166.25, |
| "completions/min_length": 111.0, |
| "epoch": 2.376068376068376, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9037185878824516, |
| "learning_rate": 6.90809614869525e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 278, |
| "step_time": 18.97715715598315 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 229.0, |
| "completions/mean_length": 157.625, |
| "completions/min_length": 91.0, |
| "epoch": 2.3846153846153846, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.88678686150806e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 279, |
| "step_time": 18.94178837002255 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 348.0, |
| "completions/mean_length": 188.1666717529297, |
| "completions/min_length": 122.0, |
| "epoch": 2.393162393162393, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1394174148350187, |
| "learning_rate": 6.865437538100456e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 280, |
| "step_time": 19.54470542888157 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 275.0, |
| "completions/mean_length": 179.7916717529297, |
| "completions/min_length": 126.0, |
| "epoch": 2.4017094017094016, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.391896025058249, |
| "learning_rate": 6.844048631489277e-07, |
| "loss": 7.450580596923828e-09, |
| "reward": 0.75, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 281, |
| "step_time": 18.796172738075256 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 192.5, |
| "completions/min_length": 128.0, |
| "epoch": 2.41025641025641, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.790125082489324, |
| "learning_rate": 6.822620595531285e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.0833333358168602, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.0833333358168602, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 282, |
| "step_time": 22.95623058313504 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 176.0, |
| "completions/mean_length": 123.54167175292969, |
| "completions/min_length": 86.0, |
| "epoch": 2.4188034188034186, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2852447069226463, |
| "learning_rate": 6.80115388491354e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 283, |
| "step_time": 17.760611035861075 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 306.0, |
| "completions/mean_length": 186.08334350585938, |
| "completions/min_length": 107.0, |
| "epoch": 2.427350427350427, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5727587522600355, |
| "learning_rate": 6.779648955143753e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 284, |
| "step_time": 16.915416153147817 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 228.0, |
| "completions/mean_length": 163.25, |
| "completions/min_length": 100.0, |
| "epoch": 2.435897435897436, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.3543506719242644, |
| "learning_rate": 6.75810626254061e-07, |
| "loss": 2.9802322387695312e-08, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 285, |
| "step_time": 19.17064887401648 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 254.0, |
| "completions/mean_length": 166.2916717529297, |
| "completions/min_length": 122.0, |
| "epoch": 2.4444444444444446, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8235138355491696, |
| "learning_rate": 6.7365262642241e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 286, |
| "step_time": 17.574008862953633 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 256.625, |
| "completions/min_length": 133.0, |
| "epoch": 2.452991452991453, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.771324501565867, |
| "learning_rate": 6.714909418105815e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 287, |
| "step_time": 19.173621407011524 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 201.125, |
| "completions/min_length": 128.0, |
| "epoch": 2.4615384615384617, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.598545230728024, |
| "learning_rate": 6.693256182879223e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.0416666679084301, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.0416666679084301, |
| "rewards/DirectReward/std": 0.20412413775920868, |
| "step": 288, |
| "step_time": 22.19392727711238 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 335.0, |
| "completions/mean_length": 146.45834350585938, |
| "completions/min_length": 84.0, |
| "epoch": 2.47008547008547, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.671567018009947e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 289, |
| "step_time": 19.792234638007358 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 220.0, |
| "completions/mean_length": 159.45834350585938, |
| "completions/min_length": 107.0, |
| "epoch": 2.4786324786324787, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5209409875428634, |
| "learning_rate": 6.64984238372601e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 290, |
| "step_time": 16.307848608121276 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 211.0, |
| "completions/mean_length": 155.58334350585938, |
| "completions/min_length": 124.0, |
| "epoch": 2.4871794871794872, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.628082741008067e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 291, |
| "step_time": 17.847699223086238 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 325.0, |
| "completions/mean_length": 181.20834350585938, |
| "completions/min_length": 121.0, |
| "epoch": 2.4957264957264957, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5925763596220097, |
| "learning_rate": 6.606288551579629e-07, |
| "loss": 3.725290298461914e-08, |
| "reward": 0.75, |
| "reward_std": 0.34503278136253357, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 292, |
| "step_time": 20.57742049708031 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 299.0, |
| "completions/mean_length": 162.58334350585938, |
| "completions/min_length": 113.0, |
| "epoch": 2.5042735042735043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.584460277897261e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 293, |
| "step_time": 19.152367496863008 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 314.0, |
| "completions/mean_length": 164.2916717529297, |
| "completions/min_length": 113.0, |
| "epoch": 2.5128205128205128, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.348101050120348, |
| "learning_rate": 6.562598383140771e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 294, |
| "step_time": 20.277267024153844 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 276.0, |
| "completions/mean_length": 155.25, |
| "completions/min_length": 107.0, |
| "epoch": 2.5213675213675213, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.509930821421422, |
| "learning_rate": 6.540703331203382e-07, |
| "loss": 0.0, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 295, |
| "step_time": 18.286975691094995 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 278.0, |
| "completions/mean_length": 150.08334350585938, |
| "completions/min_length": 99.0, |
| "epoch": 2.52991452991453, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.0525807806902003, |
| "learning_rate": 6.518775586681886e-07, |
| "loss": 0.0, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 296, |
| "step_time": 19.505128348013386 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 265.0, |
| "completions/mean_length": 166.4166717529297, |
| "completions/min_length": 109.0, |
| "epoch": 2.5384615384615383, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2809576046555646, |
| "learning_rate": 6.496815614866791e-07, |
| "loss": 0.0, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 297, |
| "step_time": 17.629483388969675 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 337.0, |
| "completions/mean_length": 168.7916717529297, |
| "completions/min_length": 95.0, |
| "epoch": 2.547008547008547, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1280417900881434, |
| "learning_rate": 6.474823881732439e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.125, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.125, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 298, |
| "step_time": 15.805983386002481 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 374.0, |
| "completions/mean_length": 178.58334350585938, |
| "completions/min_length": 124.0, |
| "epoch": 2.5555555555555554, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.546047565534918, |
| "learning_rate": 6.452800853927127e-07, |
| "loss": -2.4835269840650653e-08, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 299, |
| "step_time": 16.266180227976292 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 188.2916717529297, |
| "completions/min_length": 98.0, |
| "epoch": 2.564102564102564, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.623523536724057, |
| "learning_rate": 6.430746998763203e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 300, |
| "step_time": 21.806293641915545 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 375.0, |
| "completions/mean_length": 157.9166717529297, |
| "completions/min_length": 93.0, |
| "epoch": 2.5726495726495724, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.779692672609785, |
| "learning_rate": 6.408662784207149e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 301, |
| "step_time": 19.993974322918802 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 195.0, |
| "completions/mean_length": 141.95834350585938, |
| "completions/min_length": 93.0, |
| "epoch": 2.5811965811965814, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5380971894548003, |
| "learning_rate": 6.386548678869643e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 302, |
| "step_time": 17.00675908010453 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 212.0, |
| "completions/mean_length": 146.0, |
| "completions/min_length": 116.0, |
| "epoch": 2.58974358974359, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.566418869745, |
| "learning_rate": 6.364405151995636e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 303, |
| "step_time": 16.931453683180735 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 235.0, |
| "completions/mean_length": 158.2916717529297, |
| "completions/min_length": 117.0, |
| "epoch": 2.5982905982905984, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.197877611164594, |
| "learning_rate": 6.34223267345437e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 304, |
| "step_time": 14.911087288055569 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 257.0, |
| "completions/mean_length": 168.58334350585938, |
| "completions/min_length": 114.0, |
| "epoch": 2.606837606837607, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.508090890221131, |
| "learning_rate": 6.320031713729428e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 305, |
| "step_time": 18.47922940715216 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 272.0, |
| "completions/mean_length": 148.375, |
| "completions/min_length": 95.0, |
| "epoch": 2.6153846153846154, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.29780274390874e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 306, |
| "step_time": 14.662472844822332 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 245.0, |
| "completions/mean_length": 153.2916717529297, |
| "completions/min_length": 96.0, |
| "epoch": 2.623931623931624, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.895703749568616, |
| "learning_rate": 6.275546235674588e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 307, |
| "step_time": 18.098707044962794 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 396.0, |
| "completions/mean_length": 164.875, |
| "completions/min_length": 104.0, |
| "epoch": 2.6324786324786325, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7435817295428993, |
| "learning_rate": 6.253262661293602e-07, |
| "loss": 2.2351741790771484e-08, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 308, |
| "step_time": 20.717165804002434 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 180.0, |
| "completions/mean_length": 132.7916717529297, |
| "completions/min_length": 100.0, |
| "epoch": 2.641025641025641, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7569769987486374, |
| "learning_rate": 6.230952493606733e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 309, |
| "step_time": 16.710390684893355 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 204.0, |
| "completions/mean_length": 150.2916717529297, |
| "completions/min_length": 111.0, |
| "epoch": 2.6495726495726495, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.589083769695093, |
| "learning_rate": 6.208616206019223e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 310, |
| "step_time": 17.768274259055033 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 199.0, |
| "completions/mean_length": 147.0, |
| "completions/min_length": 101.0, |
| "epoch": 2.658119658119658, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.5156129596055905, |
| "learning_rate": 6.18625427249056e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 311, |
| "step_time": 17.14775848388672 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 203.08334350585938, |
| "completions/min_length": 92.0, |
| "epoch": 2.6666666666666665, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.5710239566870254, |
| "learning_rate": 6.163867167524418e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 312, |
| "step_time": 22.84569594101049 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 237.0, |
| "completions/mean_length": 138.1666717529297, |
| "completions/min_length": 99.0, |
| "epoch": 2.6752136752136755, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.720727346476739, |
| "learning_rate": 6.14145536615859e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 313, |
| "step_time": 17.79962663492188 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 481.0, |
| "completions/mean_length": 178.75, |
| "completions/min_length": 103.0, |
| "epoch": 2.683760683760684, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.598071042140844, |
| "learning_rate": 6.119019343954913e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.45032864809036255, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 314, |
| "step_time": 21.33371075009927 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/mean_length": 132.375, |
| "completions/min_length": 81.0, |
| "epoch": 2.6923076923076925, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.5085340294486715, |
| "learning_rate": 6.096559576989165e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 315, |
| "step_time": 16.568679050076753 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 210.0, |
| "completions/mean_length": 128.33334350585938, |
| "completions/min_length": 96.0, |
| "epoch": 2.700854700854701, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.074076541840977e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 316, |
| "step_time": 16.134605559986085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 374.0, |
| "completions/mean_length": 162.0416717529297, |
| "completions/min_length": 94.0, |
| "epoch": 2.7094017094017095, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 6.05157071558371e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 317, |
| "step_time": 18.996667401865125 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 303.0, |
| "completions/mean_length": 155.5, |
| "completions/min_length": 93.0, |
| "epoch": 2.717948717948718, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.9910042282278306, |
| "learning_rate": 6.029042575774333e-07, |
| "loss": 2.9802322387695312e-08, |
| "reward": 0.5, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 318, |
| "step_time": 19.583524979883805 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 351.0, |
| "completions/mean_length": 167.4166717529297, |
| "completions/min_length": 100.0, |
| "epoch": 2.7264957264957266, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.218758744179187, |
| "learning_rate": 6.0064926004433e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 319, |
| "step_time": 20.15218091197312 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 305.0, |
| "completions/mean_length": 139.5, |
| "completions/min_length": 83.0, |
| "epoch": 2.735042735042735, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.0844255031722474, |
| "learning_rate": 5.983921268084393e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 320, |
| "step_time": 19.270107568008825 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 232.0, |
| "completions/mean_length": 142.375, |
| "completions/min_length": 93.0, |
| "epoch": 2.7435897435897436, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.157501150388822, |
| "learning_rate": 5.96132905764457e-07, |
| "loss": 0.0, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 321, |
| "step_time": 15.291413035942242 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 217.33334350585938, |
| "completions/min_length": 110.0, |
| "epoch": 2.752136752136752, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5406529857800297, |
| "learning_rate": 5.938716448513817e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 322, |
| "step_time": 23.075299581047148 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 265.0, |
| "completions/mean_length": 161.08334350585938, |
| "completions/min_length": 101.0, |
| "epoch": 2.7606837606837606, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7879501562261493, |
| "learning_rate": 5.916083920514958e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.125, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.125, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 323, |
| "step_time": 17.810189850162715 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 338.0, |
| "completions/mean_length": 141.1666717529297, |
| "completions/min_length": 102.0, |
| "epoch": 2.769230769230769, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9936218898505573, |
| "learning_rate": 5.893431953893482e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 324, |
| "step_time": 20.820132290013134 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 491.0, |
| "completions/mean_length": 159.5, |
| "completions/min_length": 98.0, |
| "epoch": 2.7777777777777777, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.870761029307351e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 325, |
| "step_time": 19.16246618097648 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 269.0, |
| "completions/mean_length": 163.20834350585938, |
| "completions/min_length": 105.0, |
| "epoch": 2.786324786324786, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.3480320092235387, |
| "learning_rate": 5.848071627816803e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 326, |
| "step_time": 18.52779599907808 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 237.0, |
| "completions/mean_length": 142.625, |
| "completions/min_length": 91.0, |
| "epoch": 2.7948717948717947, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4527583912891404, |
| "learning_rate": 5.825364230874139e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 327, |
| "step_time": 17.36012075119652 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 196.95834350585938, |
| "completions/min_length": 94.0, |
| "epoch": 2.8034188034188032, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.977130245682578, |
| "learning_rate": 5.802639320313513e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.46854168176651, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 328, |
| "step_time": 21.54533420293592 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 183.125, |
| "completions/min_length": 100.0, |
| "epoch": 2.8119658119658117, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5741019159139036, |
| "learning_rate": 5.779897378340704e-07, |
| "loss": 0.0, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 329, |
| "step_time": 20.419018008047715 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 464.0, |
| "completions/mean_length": 150.95834350585938, |
| "completions/min_length": 92.0, |
| "epoch": 2.8205128205128203, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.318746278008293, |
| "learning_rate": 5.757138887522883e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.4082186222076416, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 330, |
| "step_time": 19.554127152077854 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 198.0, |
| "completions/mean_length": 134.9166717529297, |
| "completions/min_length": 101.0, |
| "epoch": 2.8290598290598292, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 5.14453833689308, |
| "learning_rate": 5.73436433077838e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 331, |
| "step_time": 16.836927075870335 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 203.0, |
| "completions/mean_length": 150.0, |
| "completions/min_length": 103.0, |
| "epoch": 2.8376068376068377, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.641516845405126, |
| "learning_rate": 5.711574191366427e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 332, |
| "step_time": 17.73256492591463 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 171.45834350585938, |
| "completions/min_length": 74.0, |
| "epoch": 2.8461538461538463, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 5.021217585927427, |
| "learning_rate": 5.688768952876909e-07, |
| "loss": 0.0, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.48112308979034424, |
| "rewards/DirectReward/mean": 0.4583333432674408, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 333, |
| "step_time": 23.436607328942046 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 232.0, |
| "completions/mean_length": 133.125, |
| "completions/min_length": 90.0, |
| "epoch": 2.8547008547008548, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.794112764599765, |
| "learning_rate": 5.665949099220109e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 334, |
| "step_time": 19.628532842034474 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 322.0, |
| "completions/mean_length": 145.0416717529297, |
| "completions/min_length": 85.0, |
| "epoch": 2.8632478632478633, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.706242097019197, |
| "learning_rate": 5.643115114616425e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 335, |
| "step_time": 18.39846288296394 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 380.0, |
| "completions/mean_length": 153.25, |
| "completions/min_length": 95.0, |
| "epoch": 2.871794871794872, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.620267483586104e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 336, |
| "step_time": 17.08073847205378 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 395.0, |
| "completions/mean_length": 136.08334350585938, |
| "completions/min_length": 75.0, |
| "epoch": 2.8803418803418803, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.3579831449688, |
| "learning_rate": 5.597406690938968e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 337, |
| "step_time": 20.46336353314109 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 180.0, |
| "completions/mean_length": 134.08334350585938, |
| "completions/min_length": 89.0, |
| "epoch": 2.888888888888889, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.6545959384962865, |
| "learning_rate": 5.574533221764108e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 338, |
| "step_time": 17.065708016976714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 168.0, |
| "completions/mean_length": 117.79167175292969, |
| "completions/min_length": 94.0, |
| "epoch": 2.8974358974358974, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.860543628870516, |
| "learning_rate": 5.55164756141961e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 339, |
| "step_time": 15.366660661995411 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 159.20834350585938, |
| "completions/min_length": 81.0, |
| "epoch": 2.905982905982906, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.528750195522243e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 340, |
| "step_time": 19.027446056017652 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 178.6666717529297, |
| "completions/min_length": 92.0, |
| "epoch": 2.9145299145299144, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.0176878226038437, |
| "learning_rate": 5.505841609937161e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 341, |
| "step_time": 22.617938708979636 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 209.0, |
| "completions/mean_length": 134.75, |
| "completions/min_length": 97.0, |
| "epoch": 2.9230769230769234, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.059091688538997, |
| "learning_rate": 5.482922290767588e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 342, |
| "step_time": 17.256556726060808 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 178.0, |
| "completions/mean_length": 134.0416717529297, |
| "completions/min_length": 96.0, |
| "epoch": 2.931623931623932, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.459992724344515e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 343, |
| "step_time": 17.68903303705156 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 166.5, |
| "completions/min_length": 97.0, |
| "epoch": 2.9401709401709404, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.709862543830365, |
| "learning_rate": 5.437053397216364e-07, |
| "loss": 0.0, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.4082186222076416, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 344, |
| "step_time": 18.370430425042287 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 216.0, |
| "completions/mean_length": 141.875, |
| "completions/min_length": 82.0, |
| "epoch": 2.948717948717949, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7590234236533338, |
| "learning_rate": 5.414104796138672e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 345, |
| "step_time": 19.407583627151325 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 164.70834350585938, |
| "completions/min_length": 100.0, |
| "epoch": 2.9572649572649574, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.811148669626966, |
| "learning_rate": 5.39114740806377e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 346, |
| "step_time": 20.784914833959192 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 228.75, |
| "completions/min_length": 128.0, |
| "epoch": 2.965811965811966, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9341989936063437, |
| "learning_rate": 5.368181720130434e-07, |
| "loss": 0.0, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 347, |
| "step_time": 19.270138487918302 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 155.6666717529297, |
| "completions/min_length": 69.0, |
| "epoch": 2.9743589743589745, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.4296318552371337, |
| "learning_rate": 5.345208219653561e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.75, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 348, |
| "step_time": 22.380040783202276 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 233.0, |
| "completions/mean_length": 139.0, |
| "completions/min_length": 83.0, |
| "epoch": 2.982905982905983, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.322227394113825e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 349, |
| "step_time": 18.35665961704217 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 167.0, |
| "completions/mean_length": 135.5, |
| "completions/min_length": 100.0, |
| "epoch": 2.9914529914529915, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.299239731147331e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 350, |
| "step_time": 18.616521089104936 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 198.45834350585938, |
| "completions/min_length": 113.0, |
| "epoch": 3.0, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.974169076820545, |
| "learning_rate": 5.276245718535269e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.375, |
| "reward_std": 0.47419947385787964, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 351, |
| "step_time": 23.222488602157682 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 177.0, |
| "completions/mean_length": 124.875, |
| "completions/min_length": 81.0, |
| "epoch": 3.0085470085470085, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.242539658970827, |
| "learning_rate": 5.253245844193564e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 352, |
| "step_time": 16.56993775581941 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 185.0416717529297, |
| "completions/min_length": 114.0, |
| "epoch": 3.017094017094017, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.9916951093471034, |
| "learning_rate": 5.230240596162522e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 353, |
| "step_time": 23.40085698105395 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 151.0, |
| "completions/mean_length": 122.20833587646484, |
| "completions/min_length": 73.0, |
| "epoch": 3.0256410256410255, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5.207230462596478e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 354, |
| "step_time": 14.75097723887302 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 330.0, |
| "completions/mean_length": 149.95834350585938, |
| "completions/min_length": 91.0, |
| "epoch": 3.034188034188034, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.524289180218499, |
| "learning_rate": 5.18421593175343e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.5, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 355, |
| "step_time": 19.294492562068626 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 338.0, |
| "completions/mean_length": 139.25, |
| "completions/min_length": 100.0, |
| "epoch": 3.0427350427350426, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.091918773847688, |
| "learning_rate": 5.161197491984684e-07, |
| "loss": 0.0, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 356, |
| "step_time": 15.501012190943584 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 224.0, |
| "completions/mean_length": 153.2916717529297, |
| "completions/min_length": 109.0, |
| "epoch": 3.051282051282051, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.732965201316345, |
| "learning_rate": 5.138175631724494e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 357, |
| "step_time": 17.439264287939295 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 328.0, |
| "completions/mean_length": 143.9166717529297, |
| "completions/min_length": 90.0, |
| "epoch": 3.0598290598290596, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.1612567505353786, |
| "learning_rate": 5.11515083947969e-07, |
| "loss": 0.0, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 358, |
| "step_time": 17.235377665841952 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 214.0, |
| "completions/mean_length": 131.25, |
| "completions/min_length": 74.0, |
| "epoch": 3.0683760683760686, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9650158020435753, |
| "learning_rate": 5.092123603819317e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 359, |
| "step_time": 17.85052488790825 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 239.625, |
| "completions/min_length": 90.0, |
| "epoch": 3.076923076923077, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5306761034159684, |
| "learning_rate": 5.069094413364271e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 360, |
| "step_time": 24.72668578592129 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 329.0, |
| "completions/mean_length": 167.0416717529297, |
| "completions/min_length": 102.0, |
| "epoch": 3.0854700854700856, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.81140453812335, |
| "learning_rate": 5.046063756776925e-07, |
| "loss": 7.450580596923828e-09, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 361, |
| "step_time": 19.851621811045334 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 510.0, |
| "completions/mean_length": 170.875, |
| "completions/min_length": 96.0, |
| "epoch": 3.094017094017094, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7275424224023537, |
| "learning_rate": 5.023032122750759e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.75, |
| "reward_std": 0.34503278136253357, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 362, |
| "step_time": 22.81348634394817 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 276.0, |
| "completions/mean_length": 161.875, |
| "completions/min_length": 107.0, |
| "epoch": 3.1025641025641026, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.505065221231013, |
| "learning_rate": 5e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.3900056481361389, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 363, |
| "step_time": 19.883760500932112 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 213.0, |
| "completions/mean_length": 142.4166717529297, |
| "completions/min_length": 106.0, |
| "epoch": 3.111111111111111, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.976967877249242e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 364, |
| "step_time": 18.80104944598861 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 397.0, |
| "completions/mean_length": 174.5, |
| "completions/min_length": 89.0, |
| "epoch": 3.1196581196581197, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.440319928973018, |
| "learning_rate": 4.953936243223076e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 365, |
| "step_time": 18.293126181932166 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 188.0, |
| "completions/mean_length": 138.5416717529297, |
| "completions/min_length": 70.0, |
| "epoch": 3.128205128205128, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.93090558663573e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 366, |
| "step_time": 17.374381768051535 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 205.0416717529297, |
| "completions/min_length": 117.0, |
| "epoch": 3.1367521367521367, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.372349556106545, |
| "learning_rate": 4.907876396180684e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 367, |
| "step_time": 22.898386192973703 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 238.0, |
| "completions/mean_length": 138.5, |
| "completions/min_length": 93.0, |
| "epoch": 3.1452991452991452, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.884849160520311e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 368, |
| "step_time": 18.615978034911677 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 223.75, |
| "completions/min_length": 106.0, |
| "epoch": 3.1538461538461537, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.500549360179753, |
| "learning_rate": 4.861824368275507e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.25, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.25, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 369, |
| "step_time": 23.08964082505554 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 182.83334350585938, |
| "completions/min_length": 93.0, |
| "epoch": 3.1623931623931623, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.0329007067042153, |
| "learning_rate": 4.838802508015316e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 370, |
| "step_time": 20.362958719953895 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 184.0, |
| "completions/mean_length": 145.58334350585938, |
| "completions/min_length": 104.0, |
| "epoch": 3.1709401709401708, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7695767997626883, |
| "learning_rate": 4.815784068246571e-07, |
| "loss": 0.0, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 371, |
| "step_time": 17.830423589097336 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 216.0, |
| "completions/mean_length": 149.0416717529297, |
| "completions/min_length": 101.0, |
| "epoch": 3.1794871794871793, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.1354618341868408, |
| "learning_rate": 4.792769537403523e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 372, |
| "step_time": 17.98223467892967 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 198.0, |
| "completions/mean_length": 136.375, |
| "completions/min_length": 103.0, |
| "epoch": 3.1880341880341883, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.769759403837479e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 373, |
| "step_time": 17.050103277899325 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 173.4166717529297, |
| "completions/min_length": 104.0, |
| "epoch": 3.1965811965811968, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.5080208589883517, |
| "learning_rate": 4.746754155806437e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 374, |
| "step_time": 22.688060910906643 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 199.0, |
| "completions/mean_length": 130.9166717529297, |
| "completions/min_length": 101.0, |
| "epoch": 3.2051282051282053, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.723754281464732e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 375, |
| "step_time": 17.735914502991363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 200.0, |
| "completions/mean_length": 148.875, |
| "completions/min_length": 116.0, |
| "epoch": 3.213675213675214, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.9750419363898835, |
| "learning_rate": 4.7007602688526687e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.625, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 376, |
| "step_time": 17.581162279006094 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 486.0, |
| "completions/mean_length": 183.33334350585938, |
| "completions/min_length": 103.0, |
| "epoch": 3.2222222222222223, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.6738424685779316, |
| "learning_rate": 4.6777726058861747e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 377, |
| "step_time": 22.162482955958694 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 181.1666717529297, |
| "completions/min_length": 89.0, |
| "epoch": 3.230769230769231, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.264927363170676, |
| "learning_rate": 4.654791780346439e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 378, |
| "step_time": 22.373097776900977 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 247.0, |
| "completions/mean_length": 140.20834350585938, |
| "completions/min_length": 101.0, |
| "epoch": 3.2393162393162394, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.326318532887606, |
| "learning_rate": 4.631818279869566e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 379, |
| "step_time": 19.191335865063593 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 167.0, |
| "completions/mean_length": 128.5, |
| "completions/min_length": 101.0, |
| "epoch": 3.247863247863248, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.746106259224753, |
| "learning_rate": 4.6088525919362305e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 380, |
| "step_time": 16.14285749802366 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 319.0, |
| "completions/mean_length": 153.83334350585938, |
| "completions/min_length": 99.0, |
| "epoch": 3.2564102564102564, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.585895203861328e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 381, |
| "step_time": 15.734154707984999 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 170.0, |
| "completions/mean_length": 134.875, |
| "completions/min_length": 84.0, |
| "epoch": 3.264957264957265, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.562946602783636e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 382, |
| "step_time": 16.636973245069385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 180.75, |
| "completions/min_length": 90.0, |
| "epoch": 3.2735042735042734, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 9.599384252089559, |
| "learning_rate": 4.5400072756554845e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.75, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 383, |
| "step_time": 22.220648923888803 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 210.1666717529297, |
| "completions/min_length": 102.0, |
| "epoch": 3.282051282051282, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.4903123778110516, |
| "learning_rate": 4.517077709232411e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 384, |
| "step_time": 22.93072783900425 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 224.0, |
| "completions/mean_length": 150.2916717529297, |
| "completions/min_length": 104.0, |
| "epoch": 3.2905982905982905, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.905965152446692, |
| "learning_rate": 4.4941583900628393e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 385, |
| "step_time": 14.646286322036758 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 187.0, |
| "completions/mean_length": 143.1666717529297, |
| "completions/min_length": 103.0, |
| "epoch": 3.299145299145299, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8830035077707956, |
| "learning_rate": 4.4712498044777577e-07, |
| "loss": 0.0, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 386, |
| "step_time": 17.310228287940845 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 190.0, |
| "completions/mean_length": 136.5416717529297, |
| "completions/min_length": 91.0, |
| "epoch": 3.3076923076923075, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.104826403973318, |
| "learning_rate": 4.4483524385803904e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.375, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 387, |
| "step_time": 17.342273438815027 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 151.0, |
| "completions/mean_length": 112.45833587646484, |
| "completions/min_length": 85.0, |
| "epoch": 3.316239316239316, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.4254667782358916e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 388, |
| "step_time": 16.969406741904095 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 240.0, |
| "completions/mean_length": 138.1666717529297, |
| "completions/min_length": 92.0, |
| "epoch": 3.324786324786325, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2792355795805657, |
| "learning_rate": 4.4025933090610334e-07, |
| "loss": 0.0, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 389, |
| "step_time": 17.478768409928307 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 217.0, |
| "completions/mean_length": 143.0, |
| "completions/min_length": 95.0, |
| "epoch": 3.3333333333333335, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.3797325164138964e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 390, |
| "step_time": 17.36949072498828 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 159.2916717529297, |
| "completions/min_length": 85.0, |
| "epoch": 3.341880341880342, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.3125066712037903, |
| "learning_rate": 4.356884885383577e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 391, |
| "step_time": 22.451568922027946 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 467.0, |
| "completions/mean_length": 160.08334350585938, |
| "completions/min_length": 98.0, |
| "epoch": 3.3504273504273505, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9694484954173763, |
| "learning_rate": 4.334050900779893e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 392, |
| "step_time": 22.073846000013873 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 268.0, |
| "completions/mean_length": 135.08334350585938, |
| "completions/min_length": 92.0, |
| "epoch": 3.358974358974359, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.311231047123092e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 393, |
| "step_time": 15.285164901055396 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 284.0, |
| "completions/mean_length": 141.70834350585938, |
| "completions/min_length": 90.0, |
| "epoch": 3.3675213675213675, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5431152276722173, |
| "learning_rate": 4.2884258086335745e-07, |
| "loss": 0.0, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 394, |
| "step_time": 17.467965929070488 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 180.25, |
| "completions/min_length": 105.0, |
| "epoch": 3.376068376068376, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.4239045890935547, |
| "learning_rate": 4.2656356692216217e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.1666666716337204, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.1666666716337204, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 395, |
| "step_time": 21.913937548175454 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 220.0, |
| "completions/mean_length": 145.95834350585938, |
| "completions/min_length": 117.0, |
| "epoch": 3.3846153846153846, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.242861112477118e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 396, |
| "step_time": 17.087945719016716 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 158.4166717529297, |
| "completions/min_length": 82.0, |
| "epoch": 3.393162393162393, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.3730040631988767, |
| "learning_rate": 4.2201026216592973e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 397, |
| "step_time": 18.96277489908971 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/mean_length": 134.4166717529297, |
| "completions/min_length": 89.0, |
| "epoch": 3.4017094017094016, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.197360679686488e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 398, |
| "step_time": 14.061865689931437 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 243.0, |
| "completions/mean_length": 144.25, |
| "completions/min_length": 104.0, |
| "epoch": 3.41025641025641, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7037220313575343, |
| "learning_rate": 4.174635769125861e-07, |
| "loss": -1.9868215517249155e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 399, |
| "step_time": 18.092741801869124 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 198.0, |
| "completions/mean_length": 144.875, |
| "completions/min_length": 93.0, |
| "epoch": 3.4188034188034186, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 4.1519283721831975e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 400, |
| "step_time": 17.52652409207076 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 253.0, |
| "completions/mean_length": 126.41667175292969, |
| "completions/min_length": 91.0, |
| "epoch": 3.427350427350427, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5142250193584306, |
| "learning_rate": 4.1292389706926503e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.75, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 401, |
| "step_time": 18.498217853950337 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 217.4166717529297, |
| "completions/min_length": 120.0, |
| "epoch": 3.435897435897436, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.49513683497457, |
| "learning_rate": 4.106568046106519e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.46854168176651, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 402, |
| "step_time": 20.85349760693498 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 377.0, |
| "completions/mean_length": 142.125, |
| "completions/min_length": 87.0, |
| "epoch": 3.4444444444444446, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7142292243578594, |
| "learning_rate": 4.083916079485043e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 403, |
| "step_time": 21.55698793497868 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/mean_length": 140.0416717529297, |
| "completions/min_length": 108.0, |
| "epoch": 3.452991452991453, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.3207741192078295, |
| "learning_rate": 4.0612835514861845e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 404, |
| "step_time": 17.389766878215596 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 201.0, |
| "completions/mean_length": 130.70834350585938, |
| "completions/min_length": 98.0, |
| "epoch": 3.4615384615384617, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.140953616614906, |
| "learning_rate": 4.0386709423554305e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 405, |
| "step_time": 16.62493245699443 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 151.0, |
| "completions/min_length": 83.0, |
| "epoch": 3.47008547008547, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.35637850283364, |
| "learning_rate": 4.0160787319156076e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 406, |
| "step_time": 22.506701997015625 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 149.0, |
| "completions/mean_length": 108.20833587646484, |
| "completions/min_length": 78.0, |
| "epoch": 3.4786324786324787, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.9935073995566987e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 407, |
| "step_time": 16.869847586145625 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 186.33334350585938, |
| "completions/min_length": 111.0, |
| "epoch": 3.4871794871794872, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.904655780360655, |
| "learning_rate": 3.9709574242256657e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.3506905436515808, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 408, |
| "step_time": 24.33419958408922 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 306.0, |
| "completions/mean_length": 156.7916717529297, |
| "completions/min_length": 99.0, |
| "epoch": 3.4957264957264957, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.023180951795853, |
| "learning_rate": 3.94842928441629e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 409, |
| "step_time": 19.187930868938565 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 266.0, |
| "completions/mean_length": 131.2916717529297, |
| "completions/min_length": 82.0, |
| "epoch": 3.5042735042735043, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.1059067860361407, |
| "learning_rate": 3.9259234581590224e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 410, |
| "step_time": 18.81188228307292 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 160.0, |
| "completions/mean_length": 112.875, |
| "completions/min_length": 79.0, |
| "epoch": 3.5128205128205128, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.001482715877476, |
| "learning_rate": 3.903440423010834e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 411, |
| "step_time": 17.691419971175492 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 209.0, |
| "completions/mean_length": 126.25, |
| "completions/min_length": 80.0, |
| "epoch": 3.5213675213675213, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.0620129925391404, |
| "learning_rate": 3.8809806560450863e-07, |
| "loss": 0.0, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 412, |
| "step_time": 15.955848877085373 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 337.0, |
| "completions/mean_length": 152.875, |
| "completions/min_length": 115.0, |
| "epoch": 3.52991452991453, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.052866466321705, |
| "learning_rate": 3.8585446338414084e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 413, |
| "step_time": 20.125778696034104 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 166.0, |
| "completions/mean_length": 113.08333587646484, |
| "completions/min_length": 95.0, |
| "epoch": 3.5384615384615383, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.5677193872219557, |
| "learning_rate": 3.836132832475582e-07, |
| "loss": 0.0, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 414, |
| "step_time": 16.652583630988374 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 245.0, |
| "completions/mean_length": 148.7916717529297, |
| "completions/min_length": 112.0, |
| "epoch": 3.547008547008547, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.769170215956812, |
| "learning_rate": 3.813745727509439e-07, |
| "loss": 2.2351741790771484e-08, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 415, |
| "step_time": 17.710906466003507 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 412.0, |
| "completions/mean_length": 173.1666717529297, |
| "completions/min_length": 104.0, |
| "epoch": 3.5555555555555554, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.778606081125709, |
| "learning_rate": 3.791383793980776e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.3900056481361389, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 416, |
| "step_time": 16.983598599908873 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 191.20834350585938, |
| "completions/min_length": 99.0, |
| "epoch": 3.564102564102564, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.302208273245337, |
| "learning_rate": 3.769047506393266e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.46288391947746277, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 417, |
| "step_time": 20.18035388807766 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 167.0, |
| "completions/mean_length": 117.75, |
| "completions/min_length": 81.0, |
| "epoch": 3.5726495726495724, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.7467373387063964e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 418, |
| "step_time": 14.837823951151222 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 174.0, |
| "completions/mean_length": 129.6666717529297, |
| "completions/min_length": 96.0, |
| "epoch": 3.5811965811965814, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.724453764325411e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 419, |
| "step_time": 16.80981640703976 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 249.0, |
| "completions/mean_length": 150.58334350585938, |
| "completions/min_length": 81.0, |
| "epoch": 3.58974358974359, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.5541219545512415, |
| "learning_rate": 3.7021972560912595e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 420, |
| "step_time": 18.49709944310598 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 190.0, |
| "completions/mean_length": 148.875, |
| "completions/min_length": 104.0, |
| "epoch": 3.5982905982905984, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.016419210528472, |
| "learning_rate": 3.6799682862705706e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 421, |
| "step_time": 18.333493534009904 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 226.0, |
| "completions/min_length": 88.0, |
| "epoch": 3.606837606837607, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.60924753961565, |
| "learning_rate": 3.6577673265456296e-07, |
| "loss": -7.450580596923828e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 422, |
| "step_time": 22.589360862970352 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 154.0, |
| "completions/mean_length": 123.04167175292969, |
| "completions/min_length": 99.0, |
| "epoch": 3.6153846153846154, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.17052756822909, |
| "learning_rate": 3.6355948480043644e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 423, |
| "step_time": 16.74113065097481 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 300.0, |
| "completions/mean_length": 131.25, |
| "completions/min_length": 79.0, |
| "epoch": 3.623931623931624, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.302927063590001, |
| "learning_rate": 3.613451321130355e-07, |
| "loss": 0.0, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 424, |
| "step_time": 18.417079247068614 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 166.4166717529297, |
| "completions/min_length": 85.0, |
| "epoch": 3.6324786324786325, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.6711884338174174, |
| "learning_rate": 3.591337215792851e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 425, |
| "step_time": 20.04961170000024 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 183.0, |
| "completions/mean_length": 140.5416717529297, |
| "completions/min_length": 108.0, |
| "epoch": 3.641025641025641, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.096804099652478, |
| "learning_rate": 3.569253001236795e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.4166666567325592, |
| "rewards/DirectReward/std": 0.5036101937294006, |
| "step": 426, |
| "step_time": 15.926246159011498 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 191.0, |
| "completions/mean_length": 123.08333587646484, |
| "completions/min_length": 95.0, |
| "epoch": 3.6495726495726495, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.3797515370503244, |
| "learning_rate": 3.5471991460728725e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 427, |
| "step_time": 17.15187731804326 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 350.0, |
| "completions/mean_length": 137.08334350585938, |
| "completions/min_length": 87.0, |
| "epoch": 3.658119658119658, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.376101844396586, |
| "learning_rate": 3.525176118267562e-07, |
| "loss": 0.0, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 428, |
| "step_time": 19.473717151908204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 341.0, |
| "completions/mean_length": 148.08334350585938, |
| "completions/min_length": 88.0, |
| "epoch": 3.6666666666666665, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.374990545900999, |
| "learning_rate": 3.50318438513321e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 429, |
| "step_time": 18.941036444157362 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 193.0, |
| "completions/mean_length": 130.08334350585938, |
| "completions/min_length": 86.0, |
| "epoch": 3.6752136752136755, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.4017932953858856, |
| "learning_rate": 3.481224413318113e-07, |
| "loss": 0.0, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 430, |
| "step_time": 17.550284940050915 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 155.0, |
| "completions/mean_length": 107.25, |
| "completions/min_length": 61.0, |
| "epoch": 3.683760683760684, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8978221531769477, |
| "learning_rate": 3.4592966687966185e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 431, |
| "step_time": 15.126135065918788 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 336.0, |
| "completions/mean_length": 150.33334350585938, |
| "completions/min_length": 101.0, |
| "epoch": 3.6923076923076925, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.413883523039939, |
| "learning_rate": 3.437401616859229e-07, |
| "loss": 0.0, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 432, |
| "step_time": 19.96803787490353 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 429.0, |
| "completions/mean_length": 159.83334350585938, |
| "completions/min_length": 85.0, |
| "epoch": 3.700854700854701, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.6234046805882896, |
| "learning_rate": 3.415539722102739e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 433, |
| "step_time": 21.144781715935096 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 297.0, |
| "completions/mean_length": 151.70834350585938, |
| "completions/min_length": 81.0, |
| "epoch": 3.7094017094017095, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4788021701440077, |
| "learning_rate": 3.3937114484203717e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 434, |
| "step_time": 19.751611293060705 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 306.0, |
| "completions/mean_length": 156.70834350585938, |
| "completions/min_length": 88.0, |
| "epoch": 3.717948717948718, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.699606711776227, |
| "learning_rate": 3.3719172589919326e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.0416666679084301, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.0416666679084301, |
| "rewards/DirectReward/std": 0.20412413775920868, |
| "step": 435, |
| "step_time": 19.3180920179002 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 184.0, |
| "completions/mean_length": 122.79167175292969, |
| "completions/min_length": 87.0, |
| "epoch": 3.7264957264957266, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.0954838594063094, |
| "learning_rate": 3.3501576162739897e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 436, |
| "step_time": 14.890542908804491 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 181.0, |
| "completions/mean_length": 134.33334350585938, |
| "completions/min_length": 106.0, |
| "epoch": 3.735042735042735, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.048808539005244, |
| "learning_rate": 3.3284329819900526e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 437, |
| "step_time": 17.184552452992648 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 176.0, |
| "completions/mean_length": 127.66667175292969, |
| "completions/min_length": 94.0, |
| "epoch": 3.7435897435897436, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.306743817120776e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 438, |
| "step_time": 16.11688712798059 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 430.0, |
| "completions/mean_length": 153.125, |
| "completions/min_length": 73.0, |
| "epoch": 3.752136752136752, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.95137449808075, |
| "learning_rate": 3.285090581894185e-07, |
| "loss": 0.0, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 439, |
| "step_time": 20.750867604045197 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 206.0, |
| "completions/mean_length": 137.20834350585938, |
| "completions/min_length": 102.0, |
| "epoch": 3.7606837606837606, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.263473735775899e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 440, |
| "step_time": 17.727359753102064 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 155.0, |
| "completions/mean_length": 123.625, |
| "completions/min_length": 100.0, |
| "epoch": 3.769230769230769, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.241893737459389e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 441, |
| "step_time": 17.526880905032158 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 212.0, |
| "completions/mean_length": 115.83333587646484, |
| "completions/min_length": 79.0, |
| "epoch": 3.7777777777777777, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.821041796894272, |
| "learning_rate": 3.2203510448562465e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 442, |
| "step_time": 16.872752486960962 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 260.0, |
| "completions/mean_length": 154.20834350585938, |
| "completions/min_length": 86.0, |
| "epoch": 3.786324786324786, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.857762666544264, |
| "learning_rate": 3.1988461150864585e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 443, |
| "step_time": 17.028902302030474 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 176.0, |
| "completions/mean_length": 118.29167175292969, |
| "completions/min_length": 76.0, |
| "epoch": 3.7948717948717947, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.177379404468715e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 444, |
| "step_time": 17.523418460041285 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 185.0, |
| "completions/mean_length": 150.58334350585938, |
| "completions/min_length": 101.0, |
| "epoch": 3.8034188034188032, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.607047602439452, |
| "learning_rate": 3.155951368510723e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 445, |
| "step_time": 16.96721568494104 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 312.0, |
| "completions/mean_length": 133.7916717529297, |
| "completions/min_length": 78.0, |
| "epoch": 3.8119658119658117, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 5.841868782909467, |
| "learning_rate": 3.1345624618995443e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.125, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.125, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 446, |
| "step_time": 18.721564803970978 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 432.0, |
| "completions/mean_length": 166.625, |
| "completions/min_length": 99.0, |
| "epoch": 3.8205128205128203, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.417946693721914, |
| "learning_rate": 3.1132131384919395e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 447, |
| "step_time": 20.87522027292289 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 227.6666717529297, |
| "completions/min_length": 85.0, |
| "epoch": 3.8290598290598292, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.3445155560556685, |
| "learning_rate": 3.09190385130475e-07, |
| "loss": -1.4901161193847656e-08, |
| "reward": 0.1666666716337204, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.1666666716337204, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 448, |
| "step_time": 20.92181466612965 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 172.45834350585938, |
| "completions/min_length": 92.0, |
| "epoch": 3.8376068376068377, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2190526654520513, |
| "learning_rate": 3.0706350525052726e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 449, |
| "step_time": 22.882798375096172 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 211.0, |
| "completions/mean_length": 128.95834350585938, |
| "completions/min_length": 92.0, |
| "epoch": 3.8461538461538463, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 3.0494071934016736e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 450, |
| "step_time": 14.748402397148311 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 205.0, |
| "completions/mean_length": 126.95833587646484, |
| "completions/min_length": 86.0, |
| "epoch": 3.8547008547008548, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4176763416936016, |
| "learning_rate": 3.028220724433408e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 451, |
| "step_time": 19.305605474859476 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 187.20834350585938, |
| "completions/min_length": 94.0, |
| "epoch": 3.8632478632478633, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.8444049560564917, |
| "learning_rate": 3.0070760951616616e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 452, |
| "step_time": 22.430934267118573 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 220.0, |
| "completions/mean_length": 152.83334350585938, |
| "completions/min_length": 111.0, |
| "epoch": 3.871794871794872, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.985973754259815e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 453, |
| "step_time": 17.801261613843963 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 171.2916717529297, |
| "completions/min_length": 117.0, |
| "epoch": 3.8803418803418803, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.8211628366690378, |
| "learning_rate": 2.964914149503922e-07, |
| "loss": -1.7384689243726825e-08, |
| "reward": 0.5, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 454, |
| "step_time": 25.01115168700926 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 276.0, |
| "completions/mean_length": 135.45834350585938, |
| "completions/min_length": 75.0, |
| "epoch": 3.888888888888889, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.9438977277632015e-07, |
| "loss": 0.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 455, |
| "step_time": 17.199289839016274 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 199.0, |
| "completions/mean_length": 127.29167175292969, |
| "completions/min_length": 78.0, |
| "epoch": 3.8974358974358974, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.6494347191303547, |
| "learning_rate": 2.922924934990568e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.375, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 456, |
| "step_time": 17.252972137881443 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 137.0, |
| "completions/mean_length": 110.95833587646484, |
| "completions/min_length": 91.0, |
| "epoch": 3.905982905982906, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.901996216213156e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 457, |
| "step_time": 14.19789722096175 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 179.0, |
| "completions/mean_length": 119.29167175292969, |
| "completions/min_length": 83.0, |
| "epoch": 3.9145299145299144, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.35093320674611, |
| "learning_rate": 2.881112015522884e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 458, |
| "step_time": 13.880153869045898 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 406.0, |
| "completions/mean_length": 149.6666717529297, |
| "completions/min_length": 94.0, |
| "epoch": 3.9230769230769234, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.217478029459562, |
| "learning_rate": 2.8602727760670333e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 459, |
| "step_time": 20.74374296911992 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 146.20834350585938, |
| "completions/min_length": 88.0, |
| "epoch": 3.931623931623932, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 1.4708029014709398, |
| "learning_rate": 2.8394789400388326e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 460, |
| "step_time": 21.93355446588248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 164.45834350585938, |
| "completions/min_length": 95.0, |
| "epoch": 3.9401709401709404, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.3252531712348166, |
| "learning_rate": 2.8187309486680923e-07, |
| "loss": 7.450580596923828e-09, |
| "reward": 0.75, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 461, |
| "step_time": 21.634343290003017 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 166.6666717529297, |
| "completions/min_length": 96.0, |
| "epoch": 3.948717948717949, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.7980292422118277e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 462, |
| "step_time": 18.87273257295601 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 198.0, |
| "completions/mean_length": 129.08334350585938, |
| "completions/min_length": 78.0, |
| "epoch": 3.9572649572649574, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7830952989000224, |
| "learning_rate": 2.7773742599449283e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 463, |
| "step_time": 17.031618595821783 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 267.0, |
| "completions/mean_length": 150.58334350585938, |
| "completions/min_length": 82.0, |
| "epoch": 3.965811965811966, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.756766440150822e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 464, |
| "step_time": 19.411957636941224 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 373.0, |
| "completions/mean_length": 148.1666717529297, |
| "completions/min_length": 89.0, |
| "epoch": 3.9743589743589745, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.736206220112192e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 465, |
| "step_time": 20.454097653040662 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 299.0, |
| "completions/mean_length": 159.4166717529297, |
| "completions/min_length": 111.0, |
| "epoch": 3.982905982905983, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.630959625964765, |
| "learning_rate": 2.715694036101686e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 466, |
| "step_time": 19.382094020023942 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 338.0, |
| "completions/mean_length": 161.33334350585938, |
| "completions/min_length": 109.0, |
| "epoch": 3.9914529914529915, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.7409923011034945, |
| "learning_rate": 2.6952303233726627e-07, |
| "loss": -2.4835269396561444e-09, |
| "reward": 0.375, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 467, |
| "step_time": 20.00640089600347 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 168.125, |
| "completions/min_length": 113.0, |
| "epoch": 4.0, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.750731413305848, |
| "learning_rate": 2.6748155161499565e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 468, |
| "step_time": 21.93739076098427 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 212.0, |
| "completions/mean_length": 146.58334350585938, |
| "completions/min_length": 100.0, |
| "epoch": 4.0085470085470085, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.137608492394609, |
| "learning_rate": 2.654450047620667e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.75, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 469, |
| "step_time": 17.470795248169452 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 209.0, |
| "completions/mean_length": 137.7916717529297, |
| "completions/min_length": 96.0, |
| "epoch": 4.017094017094017, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.6341343499249557e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 470, |
| "step_time": 17.57927440502681 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 202.625, |
| "completions/min_length": 103.0, |
| "epoch": 4.0256410256410255, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.6086916792395547, |
| "learning_rate": 2.6138688541468903e-07, |
| "loss": 1.2417634920325327e-08, |
| "reward": 0.375, |
| "reward_std": 0.3268197476863861, |
| "rewards/DirectReward/mean": 0.375, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 471, |
| "step_time": 21.17998196114786 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 177.0, |
| "completions/mean_length": 121.5, |
| "completions/min_length": 92.0, |
| "epoch": 4.034188034188034, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.878250588912733, |
| "learning_rate": 2.593653990305289e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 472, |
| "step_time": 14.470197268063203 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 167.0, |
| "completions/mean_length": 122.20833587646484, |
| "completions/min_length": 74.0, |
| "epoch": 4.042735042735043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.5734901873445956e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 473, |
| "step_time": 16.846822977997363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 271.0, |
| "completions/mean_length": 154.70834350585938, |
| "completions/min_length": 95.0, |
| "epoch": 4.051282051282051, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.553377873125782e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 474, |
| "step_time": 18.0613718139939 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 161.0, |
| "completions/mean_length": 125.83333587646484, |
| "completions/min_length": 102.0, |
| "epoch": 4.05982905982906, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.5333174744172704e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 475, |
| "step_time": 16.443684092955664 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 279.0, |
| "completions/mean_length": 144.9166717529297, |
| "completions/min_length": 102.0, |
| "epoch": 4.068376068376068, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.513309416885865e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 476, |
| "step_time": 17.501691329991445 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 406.0, |
| "completions/mean_length": 162.375, |
| "completions/min_length": 85.0, |
| "epoch": 4.076923076923077, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.009764818142699, |
| "learning_rate": 2.4933541250877374e-07, |
| "loss": 2.4835269840650653e-08, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 477, |
| "step_time": 18.572722112061456 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 217.0, |
| "completions/mean_length": 130.7916717529297, |
| "completions/min_length": 94.0, |
| "epoch": 4.085470085470085, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.449578591850929, |
| "learning_rate": 2.473452022459409e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 478, |
| "step_time": 17.756628329865634 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 192.625, |
| "completions/min_length": 91.0, |
| "epoch": 4.094017094017094, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.45360353130876e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 479, |
| "step_time": 20.55618921108544 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 324.0, |
| "completions/mean_length": 166.7916717529297, |
| "completions/min_length": 80.0, |
| "epoch": 4.102564102564102, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9259912614748846, |
| "learning_rate": 2.4338090728060805e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 480, |
| "step_time": 19.855411747004837 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 161.0, |
| "completions/mean_length": 129.45834350585938, |
| "completions/min_length": 86.0, |
| "epoch": 4.111111111111111, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.7464903801788827, |
| "learning_rate": 2.414069066975128e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 481, |
| "step_time": 16.898863672977313 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.16666666666666666, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 228.58334350585938, |
| "completions/min_length": 100.0, |
| "epoch": 4.119658119658119, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.584145865465591, |
| "learning_rate": 2.394383932684209e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 482, |
| "step_time": 23.21747530414723 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 154.4166717529297, |
| "completions/min_length": 100.0, |
| "epoch": 4.128205128205128, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.559201678992371, |
| "learning_rate": 2.3747540876373025e-07, |
| "loss": 0.0, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.39000558853149414, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 483, |
| "step_time": 21.22715016384609 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 216.0, |
| "completions/mean_length": 146.83334350585938, |
| "completions/min_length": 114.0, |
| "epoch": 4.136752136752137, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.0078690164310804, |
| "learning_rate": 2.355179948365189e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.8333333730697632, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.8333333134651184, |
| "rewards/DirectReward/std": 0.3806934952735901, |
| "step": 484, |
| "step_time": 17.033246567007154 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 190.4166717529297, |
| "completions/min_length": 122.0, |
| "epoch": 4.145299145299146, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.598529969802368, |
| "learning_rate": 2.335661930216611e-07, |
| "loss": -2.4835269840650653e-08, |
| "reward": 0.5, |
| "reward_std": 0.30860671401023865, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 485, |
| "step_time": 20.829962556017563 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 161.0, |
| "completions/mean_length": 124.66667175292969, |
| "completions/min_length": 89.0, |
| "epoch": 4.153846153846154, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.3162004473494657e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 486, |
| "step_time": 16.996208037016913 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 241.0, |
| "completions/mean_length": 153.6666717529297, |
| "completions/min_length": 97.0, |
| "epoch": 4.162393162393163, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.9235521810088576, |
| "learning_rate": 2.2967959127220137e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 487, |
| "step_time": 17.945400430820882 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 435.0, |
| "completions/mean_length": 169.75, |
| "completions/min_length": 75.0, |
| "epoch": 4.170940170940171, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.798965022527804, |
| "learning_rate": 2.2774487380841112e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 488, |
| "step_time": 21.30532844993286 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 190.0, |
| "completions/mean_length": 148.45834350585938, |
| "completions/min_length": 123.0, |
| "epoch": 4.17948717948718, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.2492970397510486, |
| "learning_rate": 2.2581593339684834e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 489, |
| "step_time": 17.706447020871565 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 207.0, |
| "completions/mean_length": 146.25, |
| "completions/min_length": 98.0, |
| "epoch": 4.188034188034188, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.2389281096820072e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 490, |
| "step_time": 17.29016253305599 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 213.0, |
| "completions/mean_length": 134.625, |
| "completions/min_length": 91.0, |
| "epoch": 4.196581196581197, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.2197554732970196e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 491, |
| "step_time": 16.668388770194724 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 371.0, |
| "completions/mean_length": 161.45834350585938, |
| "completions/min_length": 95.0, |
| "epoch": 4.205128205128205, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.2006418316426773e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 492, |
| "step_time": 20.007172477198765 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 210.0, |
| "completions/mean_length": 144.70834350585938, |
| "completions/min_length": 92.0, |
| "epoch": 4.213675213675214, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.1815875902963055e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 493, |
| "step_time": 17.81209530797787 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 146.0416717529297, |
| "completions/min_length": 86.0, |
| "epoch": 4.222222222222222, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.1083483631134543, |
| "learning_rate": 2.162593153574796e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 494, |
| "step_time": 18.16633710102178 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 200.95834350585938, |
| "completions/min_length": 101.0, |
| "epoch": 4.230769230769231, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.3542429825000286, |
| "learning_rate": 2.1436589245260372e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.75, |
| "reward_std": 0.33247750997543335, |
| "rewards/DirectReward/mean": 0.75, |
| "rewards/DirectReward/std": 0.4423258602619171, |
| "step": 495, |
| "step_time": 23.739790993975475 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 189.45834350585938, |
| "completions/min_length": 119.0, |
| "epoch": 4.239316239316239, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 4.277427767996564, |
| "learning_rate": 2.124785304920354e-07, |
| "loss": -9.934107758624577e-09, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.39000558853149414, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 496, |
| "step_time": 21.488015671959147 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/mean_length": 140.625, |
| "completions/min_length": 85.0, |
| "epoch": 4.247863247863248, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.745561721964058, |
| "learning_rate": 2.1059726952419781e-07, |
| "loss": 2.4835269396561444e-09, |
| "reward": 0.5416666865348816, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.5416666865348816, |
| "rewards/DirectReward/std": 0.5089773535728455, |
| "step": 497, |
| "step_time": 16.41732494602911 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 193.0, |
| "completions/mean_length": 129.4166717529297, |
| "completions/min_length": 83.0, |
| "epoch": 4.256410256410256, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.0872214946805626e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 498, |
| "step_time": 17.849722787970677 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 234.0, |
| "completions/mean_length": 141.9166717529297, |
| "completions/min_length": 84.0, |
| "epoch": 4.264957264957265, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.165124887084521, |
| "learning_rate": 2.0685321011227035e-07, |
| "loss": 0.0, |
| "reward": 0.7083333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.7083333134651184, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 499, |
| "step_time": 18.0575754721649 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 189.2916717529297, |
| "completions/min_length": 108.0, |
| "epoch": 4.273504273504273, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5494136303402932, |
| "learning_rate": 2.0499049111434918e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 500, |
| "step_time": 21.31685959198512 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 267.0, |
| "completions/mean_length": 144.4166717529297, |
| "completions/min_length": 92.0, |
| "epoch": 4.282051282051282, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.0313403199981123e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 501, |
| "step_time": 16.6999550210312 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 174.375, |
| "completions/min_length": 112.0, |
| "epoch": 4.2905982905982905, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 2.012838721613447e-07, |
| "loss": 0.0, |
| "reward": 0.3333333432674408, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.3333333432674408, |
| "rewards/DirectReward/std": 0.4815433919429779, |
| "step": 502, |
| "step_time": 22.52072742814198 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 176.0, |
| "completions/mean_length": 125.20833587646484, |
| "completions/min_length": 82.0, |
| "epoch": 4.299145299145299, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.9944005085797123e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 503, |
| "step_time": 16.714839345077053 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 199.0, |
| "completions/mean_length": 138.4166717529297, |
| "completions/min_length": 101.0, |
| "epoch": 4.3076923076923075, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.9760260721421424e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 504, |
| "step_time": 17.19866519398056 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 222.0, |
| "completions/mean_length": 145.375, |
| "completions/min_length": 109.0, |
| "epoch": 4.316239316239316, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.754017576256769, |
| "learning_rate": 1.957715802192677e-07, |
| "loss": 1.9868215517249155e-08, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 505, |
| "step_time": 18.35736120212823 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 202.0, |
| "completions/mean_length": 144.0416717529297, |
| "completions/min_length": 100.0, |
| "epoch": 4.3247863247863245, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.123168183099194, |
| "learning_rate": 1.9394700872616853e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5833333730697632, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.5833333134651184, |
| "rewards/DirectReward/std": 0.5036101341247559, |
| "step": 506, |
| "step_time": 18.252103168983012 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 274.0, |
| "completions/mean_length": 148.25, |
| "completions/min_length": 104.0, |
| "epoch": 4.333333333333333, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.9212893145097336e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 507, |
| "step_time": 18.468770613893867 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 302.0, |
| "completions/mean_length": 126.41667175292969, |
| "completions/min_length": 89.0, |
| "epoch": 4.3418803418803416, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.8165480552449784, |
| "learning_rate": 1.9031738697193615e-07, |
| "loss": 4.967053879312289e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 508, |
| "step_time": 18.875910815084353 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 175.4166717529297, |
| "completions/min_length": 93.0, |
| "epoch": 4.35042735042735, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.563098342762427, |
| "learning_rate": 1.8851241372868937e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 509, |
| "step_time": 22.335703095886856 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 393.0, |
| "completions/mean_length": 151.33334350585938, |
| "completions/min_length": 99.0, |
| "epoch": 4.358974358974359, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.8671405002142915e-07, |
| "loss": 0.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 510, |
| "step_time": 21.000576111022383 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 227.0, |
| "completions/mean_length": 141.9166717529297, |
| "completions/min_length": 82.0, |
| "epoch": 4.367521367521368, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.339370922442015, |
| "learning_rate": 1.8492233401010215e-07, |
| "loss": 0.0, |
| "reward": 0.625, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.625, |
| "rewards/DirectReward/std": 0.494535356760025, |
| "step": 511, |
| "step_time": 15.192490038927644 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 174.0, |
| "completions/mean_length": 142.125, |
| "completions/min_length": 109.0, |
| "epoch": 4.3760683760683765, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 4.65268990546078, |
| "learning_rate": 1.8313730371359547e-07, |
| "loss": 1.4901161193847656e-08, |
| "reward": 0.875, |
| "reward_std": 0.2721545100212097, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 512, |
| "step_time": 17.534377083880827 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 346.0, |
| "completions/mean_length": 151.375, |
| "completions/min_length": 100.0, |
| "epoch": 4.384615384615385, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.305883811312127, |
| "learning_rate": 1.8135899700893076e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.2357022613286972, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 513, |
| "step_time": 19.665166245074943 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 273.0, |
| "completions/mean_length": 157.2916717529297, |
| "completions/min_length": 102.0, |
| "epoch": 4.3931623931623935, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.7958745163045986e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 514, |
| "step_time": 16.149510270915926 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 188.875, |
| "completions/min_length": 96.0, |
| "epoch": 4.401709401709402, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.6856508196248825, |
| "learning_rate": 1.778227051690639e-07, |
| "loss": -4.967053879312289e-09, |
| "reward": 0.7916666865348816, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.7916666865348816, |
| "rewards/DirectReward/std": 0.4148510992527008, |
| "step": 515, |
| "step_time": 19.964789767051116 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 230.0, |
| "completions/mean_length": 141.375, |
| "completions/min_length": 96.0, |
| "epoch": 4.410256410256411, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.665329934342252, |
| "learning_rate": 1.7606479507135658e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9166666865348816, |
| "reward_std": 0.15430335700511932, |
| "rewards/DirectReward/mean": 0.9166666865348816, |
| "rewards/DirectReward/std": 0.28232985734939575, |
| "step": 516, |
| "step_time": 18.035405948059633 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 213.0, |
| "completions/mean_length": 139.08334350585938, |
| "completions/min_length": 112.0, |
| "epoch": 4.418803418803419, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.1189629016732128, |
| "learning_rate": 1.7431375863888898e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.9583333730697632, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.9583333134651184, |
| "rewards/DirectReward/std": 0.20412415266036987, |
| "step": 517, |
| "step_time": 16.0497131398879 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 266.0, |
| "completions/mean_length": 141.875, |
| "completions/min_length": 85.0, |
| "epoch": 4.427350427350428, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.725696330273575e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 518, |
| "step_time": 18.74942722497508 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 212.0, |
| "completions/mean_length": 136.4166717529297, |
| "completions/min_length": 104.0, |
| "epoch": 4.435897435897436, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.7083245524581662e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 519, |
| "step_time": 17.521432906156406 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 271.0, |
| "completions/mean_length": 165.375, |
| "completions/min_length": 117.0, |
| "epoch": 4.444444444444445, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 2.966248187098318, |
| "learning_rate": 1.6910226215589302e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.5, |
| "reward_std": 0.2903675436973572, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 520, |
| "step_time": 18.091461746953428 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 164.0, |
| "completions/min_length": 91.0, |
| "epoch": 4.452991452991453, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.043510123073003, |
| "learning_rate": 1.673790904710029e-07, |
| "loss": 9.934107758624577e-09, |
| "reward": 0.2916666865348816, |
| "reward_std": 0.1178511306643486, |
| "rewards/DirectReward/mean": 0.2916666567325592, |
| "rewards/DirectReward/std": 0.4643056094646454, |
| "step": 521, |
| "step_time": 19.897462859051302 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.08333333333333333, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 203.58334350585938, |
| "completions/min_length": 85.0, |
| "epoch": 4.461538461538462, |
| "frac_reward_zero_std": 0.3333333432674408, |
| "grad_norm": 3.5946919799680823, |
| "learning_rate": 1.656629767555739e-07, |
| "loss": -1.2417634920325327e-08, |
| "reward": 0.2083333432674408, |
| "reward_std": 0.29602527618408203, |
| "rewards/DirectReward/mean": 0.2083333283662796, |
| "rewards/DirectReward/std": 0.4148511290550232, |
| "step": 522, |
| "step_time": 19.856006670976058 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 277.0, |
| "completions/mean_length": 159.33334350585938, |
| "completions/min_length": 94.0, |
| "epoch": 4.47008547008547, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.639539574242687e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 523, |
| "step_time": 17.02178907999769 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 165.0, |
| "completions/mean_length": 129.4166717529297, |
| "completions/min_length": 80.0, |
| "epoch": 4.478632478632479, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.6225206874121217e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 524, |
| "step_time": 16.960435163928196 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 205.0, |
| "completions/mean_length": 142.95834350585938, |
| "completions/min_length": 97.0, |
| "epoch": 4.487179487179487, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.6055734681922222e-07, |
| "loss": 0.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 1.0, |
| "rewards/DirectReward/std": 0.0, |
| "step": 525, |
| "step_time": 17.39581682300195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.041666666666666664, |
| "completions/max_length": 512.0, |
| "completions/mean_length": 185.125, |
| "completions/min_length": 114.0, |
| "epoch": 4.495726495726496, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 2.4000306015099295, |
| "learning_rate": 1.5886982761904376e-07, |
| "loss": 0.0, |
| "reward": 0.5, |
| "reward_std": 0.17817416787147522, |
| "rewards/DirectReward/mean": 0.5, |
| "rewards/DirectReward/std": 0.5107539296150208, |
| "step": 526, |
| "step_time": 21.982849068008363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 205.0, |
| "completions/mean_length": 144.5, |
| "completions/min_length": 104.0, |
| "epoch": 4.504273504273504, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1.5718954694858456e-07, |
| "loss": 0.0, |
| "reward": 0.6666666865348816, |
| "reward_std": 0.0, |
| "rewards/DirectReward/mean": 0.6666666865348816, |
| "rewards/DirectReward/std": 0.4815434217453003, |
| "step": 527, |
| "step_time": 17.452307587023824 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 303.0, |
| "completions/mean_length": 147.33334350585938, |
| "completions/min_length": 111.0, |
| "epoch": 4.512820512820513, |
| "frac_reward_zero_std": 0.6666666865348816, |
| "grad_norm": 3.0530025635078477, |
| "learning_rate": 1.555165404621567e-07, |
| "loss": 1.7384689243726825e-08, |
| "reward": 0.875, |
| "reward_std": 0.17251639068126678, |
| "rewards/DirectReward/mean": 0.875, |
| "rewards/DirectReward/std": 0.337831974029541, |
| "step": 528, |
| "step_time": 17.04746055812575 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 704, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 7, |
| "save_steps": 88, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 8, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|