{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.049609326553392036, "eval_steps": 500, "global_step": 2000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 2.480466327669602e-05, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 902.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3785.0, "completions/mean_length": 5988.5, "completions/mean_terminated_length": 3785.0, "completions/min_length": 3785.0, "completions/min_terminated_length": 3785.0, "epoch": 4.960932655339204e-05, "grad_norm": 3.644390821456909, "learning_rate": 9.995e-07, "loss": -0.707, "num_tokens": 5573.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6710.0, "completions/max_terminated_length": 6710.0, "completions/mean_length": 6240.0, "completions/mean_terminated_length": 6240.0, "completions/min_length": 5770.0, "completions/min_terminated_length": 5770.0, "epoch": 7.441398983008806e-05, "grad_norm": 0.0, "learning_rate": 9.989999999999999e-07, "loss": 0.0, "num_tokens": 18951.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 3 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5443.0, "completions/max_terminated_length": 5443.0, "completions/mean_length": 5385.5, "completions/mean_terminated_length": 5385.5, "completions/min_length": 5328.0, "completions/min_terminated_length": 5328.0, "epoch": 9.921865310678408e-05, "grad_norm": 0.0, "learning_rate": 9.985e-07, "loss": 0.0, "num_tokens": 30708.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8176.0, "completions/mean_length": 8184.0, "completions/mean_terminated_length": 8176.0, "completions/min_length": 8176.0, "completions/min_terminated_length": 8176.0, "epoch": 0.00012402331638348009, "grad_norm": 0.0, "learning_rate": 9.98e-07, "loss": 0.0, "num_tokens": 39992.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1268.0, "completions/max_terminated_length": 1268.0, "completions/mean_length": 1156.5, "completions/mean_terminated_length": 1156.5, "completions/min_length": 1045.0, "completions/min_terminated_length": 1045.0, "epoch": 0.00014882797966017612, "grad_norm": 5.469529151916504, "learning_rate": 9.975e-07, "loss": 0.0682, "num_tokens": 43107.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6264.0, "completions/max_terminated_length": 6264.0, "completions/mean_length": 4586.5, "completions/mean_terminated_length": 4586.5, "completions/min_length": 2909.0, "completions/min_terminated_length": 2909.0, "epoch": 0.00017363264293687212, "grad_norm": 2.827345848083496, "learning_rate": 9.97e-07, "loss": 0.2586, "num_tokens": 53188.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 7 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1317.0, "completions/max_terminated_length": 1317.0, "completions/mean_length": 1168.0, "completions/mean_terminated_length": 1168.0, "completions/min_length": 1019.0, "completions/min_terminated_length": 1019.0, "epoch": 0.00019843730621356816, "grad_norm": 0.0, "learning_rate": 9.965e-07, "loss": 0.0, "num_tokens": 56524.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6492.0, "completions/max_terminated_length": 6492.0, "completions/mean_length": 4380.0, "completions/mean_terminated_length": 4380.0, "completions/min_length": 2268.0, "completions/min_terminated_length": 2268.0, "epoch": 0.00022324196949026416, "grad_norm": 3.096987247467041, "learning_rate": 9.959999999999999e-07, "loss": -0.3409, "num_tokens": 66266.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 9 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6320.0, "completions/max_terminated_length": 6320.0, "completions/mean_length": 5914.5, "completions/mean_terminated_length": 5914.5, "completions/min_length": 5509.0, "completions/min_terminated_length": 5509.0, "epoch": 0.00024804663276696017, "grad_norm": 0.0, "learning_rate": 9.955e-07, "loss": 0.0, "num_tokens": 79009.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 10 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.00027285129604365623, "grad_norm": 0.0, "learning_rate": 9.95e-07, "loss": 0.0, "num_tokens": 79907.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 11 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1817.0, "completions/max_terminated_length": 1817.0, "completions/mean_length": 1543.0, "completions/mean_terminated_length": 1543.0, "completions/min_length": 1269.0, "completions/min_terminated_length": 1269.0, "epoch": 0.00029765595932035224, "grad_norm": 0.0, "learning_rate": 9.945e-07, "loss": 0.0, "num_tokens": 83921.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 12 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6308.0, "completions/max_terminated_length": 6308.0, "completions/mean_length": 6150.0, "completions/mean_terminated_length": 6150.0, "completions/min_length": 5992.0, "completions/min_terminated_length": 5992.0, "epoch": 0.00032246062259704824, "grad_norm": 0.0, "learning_rate": 9.94e-07, "loss": 0.0, "num_tokens": 97075.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 13 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 641.0, "completions/max_terminated_length": 641.0, "completions/mean_length": 595.5, "completions/mean_terminated_length": 595.5, "completions/min_length": 550.0, "completions/min_terminated_length": 550.0, "epoch": 0.00034726528587374425, "grad_norm": 6.387450218200684, "learning_rate": 9.935e-07, "loss": -0.054, "num_tokens": 99056.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 14 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0003720699491504403, "grad_norm": 0.0, "learning_rate": 9.929999999999999e-07, "loss": 0.0, "num_tokens": 100062.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2966.0, "completions/max_terminated_length": 2966.0, "completions/mean_length": 1998.0, "completions/mean_terminated_length": 1998.0, "completions/min_length": 1030.0, "completions/min_terminated_length": 1030.0, "epoch": 0.0003968746124271363, "grad_norm": 0.0, "learning_rate": 9.925e-07, "loss": 0.0, "num_tokens": 104866.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 16 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0004216792757038323, "grad_norm": 0.0, "learning_rate": 9.92e-07, "loss": 0.0, "num_tokens": 105832.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 17 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2519.0, "completions/mean_length": 5355.5, "completions/mean_terminated_length": 2519.0, "completions/min_length": 2519.0, "completions/min_terminated_length": 2519.0, "epoch": 0.00044648393898052833, "grad_norm": 0.0, "learning_rate": 9.915e-07, "loss": 0.0, "num_tokens": 109163.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 18 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6722.0, "completions/max_terminated_length": 6722.0, "completions/mean_length": 6596.5, "completions/mean_terminated_length": 6596.5, "completions/min_length": 6471.0, "completions/min_terminated_length": 6471.0, "epoch": 0.00047128860225722433, "grad_norm": 0.0, "learning_rate": 9.91e-07, "loss": 0.0, "num_tokens": 123292.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 19 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2049.0, "completions/max_terminated_length": 2049.0, "completions/mean_length": 1481.0, "completions/mean_terminated_length": 1481.0, "completions/min_length": 913.0, "completions/min_terminated_length": 913.0, "epoch": 0.0004960932655339203, "grad_norm": 5.421756267547607, "learning_rate": 9.905e-07, "loss": 0.2712, "num_tokens": 127120.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1745.0, "completions/max_terminated_length": 1745.0, "completions/mean_length": 1336.0, "completions/mean_terminated_length": 1336.0, "completions/min_length": 927.0, "completions/min_terminated_length": 927.0, "epoch": 0.0005208979288106164, "grad_norm": 0.0, "learning_rate": 9.9e-07, "loss": 0.0, "num_tokens": 130608.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 21 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1126.0, "completions/max_terminated_length": 1126.0, "completions/mean_length": 979.5, "completions/mean_terminated_length": 979.5, "completions/min_length": 833.0, "completions/min_terminated_length": 833.0, "epoch": 0.0005457025920873125, "grad_norm": 0.0, "learning_rate": 9.895e-07, "loss": 0.0, "num_tokens": 133367.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 22 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 714.0, "completions/max_terminated_length": 714.0, "completions/mean_length": 607.5, "completions/mean_terminated_length": 607.5, "completions/min_length": 501.0, "completions/min_terminated_length": 501.0, "epoch": 0.0005705072553640084, "grad_norm": 0.0, "learning_rate": 9.89e-07, "loss": 0.0, "num_tokens": 135420.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 23 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6739.0, "completions/max_terminated_length": 6739.0, "completions/mean_length": 4964.5, "completions/mean_terminated_length": 4964.5, "completions/min_length": 3190.0, "completions/min_terminated_length": 3190.0, "epoch": 0.0005953119186407045, "grad_norm": 0.0, "learning_rate": 9.885e-07, "loss": 0.0, "num_tokens": 146265.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 24 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1664.0, "completions/max_terminated_length": 1664.0, "completions/mean_length": 1533.0, "completions/mean_terminated_length": 1533.0, "completions/min_length": 1402.0, "completions/min_terminated_length": 1402.0, "epoch": 0.0006201165819174004, "grad_norm": 0.0, "learning_rate": 9.88e-07, "loss": 0.0, "num_tokens": 150183.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 967.0, "completions/max_terminated_length": 967.0, "completions/mean_length": 887.5, "completions/mean_terminated_length": 887.5, "completions/min_length": 808.0, "completions/min_terminated_length": 808.0, "epoch": 0.0006449212451940965, "grad_norm": 0.0, "learning_rate": 9.875e-07, "loss": 0.0, "num_tokens": 152778.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 26 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3853.0, "completions/mean_length": 6022.5, "completions/mean_terminated_length": 3853.0, "completions/min_length": 3853.0, "completions/min_terminated_length": 3853.0, "epoch": 0.0006697259084707925, "grad_norm": 5.717548370361328, "learning_rate": 9.87e-07, "loss": -0.707, "num_tokens": 157447.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 27 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0006945305717474885, "grad_norm": 0.0, "learning_rate": 9.865e-07, "loss": 0.0, "num_tokens": 158499.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 28 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7724.0, "completions/mean_length": 7958.0, "completions/mean_terminated_length": 7724.0, "completions/min_length": 7724.0, "completions/min_terminated_length": 7724.0, "epoch": 0.0007193352350241846, "grad_norm": 2.7067646980285645, "learning_rate": 9.86e-07, "loss": -0.707, "num_tokens": 167169.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 29 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3242.0, "completions/mean_length": 5717.0, "completions/mean_terminated_length": 3242.0, "completions/min_length": 3242.0, "completions/min_terminated_length": 3242.0, "epoch": 0.0007441398983008806, "grad_norm": 4.041264533996582, "learning_rate": 9.855e-07, "loss": -0.707, "num_tokens": 171429.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4210.0, "completions/max_terminated_length": 4210.0, "completions/mean_length": 2931.0, "completions/mean_terminated_length": 2931.0, "completions/min_length": 1652.0, "completions/min_terminated_length": 1652.0, "epoch": 0.0007689445615775766, "grad_norm": 3.1480536460876465, "learning_rate": 9.849999999999999e-07, "loss": 0.3085, "num_tokens": 178175.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 31 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0007937492248542726, "grad_norm": 0.0, "learning_rate": 9.845e-07, "loss": 0.0, "num_tokens": 179083.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 32 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6139.0, "completions/max_terminated_length": 6139.0, "completions/mean_length": 4009.5, "completions/mean_terminated_length": 4009.5, "completions/min_length": 1880.0, "completions/min_terminated_length": 1880.0, "epoch": 0.0008185538881309686, "grad_norm": 0.0, "learning_rate": 9.84e-07, "loss": 0.0, "num_tokens": 188176.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 33 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0008433585514076646, "grad_norm": 0.0, "learning_rate": 9.835e-07, "loss": 0.0, "num_tokens": 189068.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 34 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5133.0, "completions/mean_length": 6662.5, "completions/mean_terminated_length": 5133.0, "completions/min_length": 5133.0, "completions/min_terminated_length": 5133.0, "epoch": 0.0008681632146843607, "grad_norm": 2.9954445362091064, "learning_rate": 9.83e-07, "loss": -0.707, "num_tokens": 195053.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1908.0, "completions/max_terminated_length": 1908.0, "completions/mean_length": 1699.0, "completions/mean_terminated_length": 1699.0, "completions/min_length": 1490.0, "completions/min_terminated_length": 1490.0, "epoch": 0.0008929678779610567, "grad_norm": 0.0, "learning_rate": 9.825e-07, "loss": 0.0, "num_tokens": 199655.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 36 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4995.0, "completions/max_terminated_length": 4995.0, "completions/mean_length": 3523.5, "completions/mean_terminated_length": 3523.5, "completions/min_length": 2052.0, "completions/min_terminated_length": 2052.0, "epoch": 0.0009177725412377527, "grad_norm": 0.0, "learning_rate": 9.819999999999999e-07, "loss": 0.0, "num_tokens": 207560.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 37 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3870.0, "completions/mean_length": 6031.0, "completions/mean_terminated_length": 3870.0, "completions/min_length": 3870.0, "completions/min_terminated_length": 3870.0, "epoch": 0.0009425772045144487, "grad_norm": 3.577725648880005, "learning_rate": 9.815e-07, "loss": -0.707, "num_tokens": 212294.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 38 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3615.0, "completions/max_terminated_length": 3615.0, "completions/mean_length": 3044.0, "completions/mean_terminated_length": 3044.0, "completions/min_length": 2473.0, "completions/min_terminated_length": 2473.0, "epoch": 0.0009673818677911447, "grad_norm": 0.0, "learning_rate": 9.81e-07, "loss": 0.0, "num_tokens": 219216.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 39 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7579.0, "completions/mean_length": 7885.5, "completions/mean_terminated_length": 7579.0, "completions/min_length": 7579.0, "completions/min_terminated_length": 7579.0, "epoch": 0.0009921865310678407, "grad_norm": 0.0, "learning_rate": 9.805e-07, "loss": 0.0, "num_tokens": 227863.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 40 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2988.0, "completions/mean_length": 5590.0, "completions/mean_terminated_length": 2988.0, "completions/min_length": 2988.0, "completions/min_terminated_length": 2988.0, "epoch": 0.0010169911943445368, "grad_norm": 0.0, "learning_rate": 9.8e-07, "loss": 0.0, "num_tokens": 231777.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 41 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5652.0, "completions/max_terminated_length": 5652.0, "completions/mean_length": 4919.5, "completions/mean_terminated_length": 4919.5, "completions/min_length": 4187.0, "completions/min_terminated_length": 4187.0, "epoch": 0.0010417958576212328, "grad_norm": 0.0, "learning_rate": 9.795e-07, "loss": 0.0, "num_tokens": 242486.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 42 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6005.0, "completions/mean_length": 7098.5, "completions/mean_terminated_length": 6005.0, "completions/min_length": 6005.0, "completions/min_terminated_length": 6005.0, "epoch": 0.0010666005208979288, "grad_norm": 0.0, "learning_rate": 9.789999999999999e-07, "loss": 0.0, "num_tokens": 249431.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 43 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5316.0, "completions/max_terminated_length": 5316.0, "completions/mean_length": 5076.0, "completions/mean_terminated_length": 5076.0, "completions/min_length": 4836.0, "completions/min_terminated_length": 4836.0, "epoch": 0.001091405184174625, "grad_norm": 2.510676860809326, "learning_rate": 9.785e-07, "loss": -0.0334, "num_tokens": 260511.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 44 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4430.0, "completions/max_terminated_length": 4430.0, "completions/mean_length": 3278.0, "completions/mean_terminated_length": 3278.0, "completions/min_length": 2126.0, "completions/min_terminated_length": 2126.0, "epoch": 0.0011162098474513209, "grad_norm": 0.0, "learning_rate": 9.78e-07, "loss": 0.0, "num_tokens": 267947.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 45 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0011410145107280168, "grad_norm": 0.0, "learning_rate": 9.775e-07, "loss": 0.0, "num_tokens": 268929.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 46 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2957.0, "completions/mean_length": 5574.5, "completions/mean_terminated_length": 2957.0, "completions/min_length": 2957.0, "completions/min_terminated_length": 2957.0, "epoch": 0.0011658191740047128, "grad_norm": 5.349734306335449, "learning_rate": 9.77e-07, "loss": -0.707, "num_tokens": 272902.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 47 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6522.0, "completions/max_terminated_length": 6522.0, "completions/mean_length": 4735.5, "completions/mean_terminated_length": 4735.5, "completions/min_length": 2949.0, "completions/min_terminated_length": 2949.0, "epoch": 0.001190623837281409, "grad_norm": 0.0, "learning_rate": 9.765e-07, "loss": 0.0, "num_tokens": 283281.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 48 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3149.0, "completions/max_terminated_length": 3149.0, "completions/mean_length": 2540.0, "completions/mean_terminated_length": 2540.0, "completions/min_length": 1931.0, "completions/min_terminated_length": 1931.0, "epoch": 0.001215428500558105, "grad_norm": 3.356837272644043, "learning_rate": 9.759999999999998e-07, "loss": -0.1695, "num_tokens": 289257.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 49 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0012402331638348009, "grad_norm": 0.0, "learning_rate": 9.755e-07, "loss": 0.0, "num_tokens": 290127.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 50 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3781.0, "completions/mean_length": 5986.5, "completions/mean_terminated_length": 3781.0, "completions/min_length": 3781.0, "completions/min_terminated_length": 3781.0, "epoch": 0.001265037827111497, "grad_norm": 0.0, "learning_rate": 9.75e-07, "loss": 0.0, "num_tokens": 294940.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 51 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.001289842490388193, "grad_norm": 0.0, "learning_rate": 9.745e-07, "loss": 0.0, "num_tokens": 295940.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 52 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6284.0, "completions/max_terminated_length": 6284.0, "completions/mean_length": 5437.0, "completions/mean_terminated_length": 5437.0, "completions/min_length": 4590.0, "completions/min_terminated_length": 4590.0, "epoch": 0.001314647153664889, "grad_norm": 2.375363826751709, "learning_rate": 9.74e-07, "loss": 0.1101, "num_tokens": 307718.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 53 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.001339451816941585, "grad_norm": 0.0, "learning_rate": 9.735e-07, "loss": 0.0, "num_tokens": 308660.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 54 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5803.0, "completions/mean_length": 6997.5, "completions/mean_terminated_length": 5803.0, "completions/min_length": 5803.0, "completions/min_terminated_length": 5803.0, "epoch": 0.001364256480218281, "grad_norm": 3.4605469703674316, "learning_rate": 9.729999999999998e-07, "loss": -0.707, "num_tokens": 315427.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 55 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.001389061143494977, "grad_norm": 0.0, "learning_rate": 9.725e-07, "loss": 0.0, "num_tokens": 316305.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 56 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0014138658067716732, "grad_norm": 0.0, "learning_rate": 9.72e-07, "loss": 0.0, "num_tokens": 317501.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 57 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0014386704700483691, "grad_norm": 0.0, "learning_rate": 9.715e-07, "loss": 0.0, "num_tokens": 318619.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 58 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.001463475133325065, "grad_norm": 0.0, "learning_rate": 9.709999999999999e-07, "loss": 0.0, "num_tokens": 319439.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 59 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0014882797966017612, "grad_norm": 0.0, "learning_rate": 9.705e-07, "loss": 0.0, "num_tokens": 320285.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 60 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6577.0, "completions/mean_length": 7384.5, "completions/mean_terminated_length": 6577.0, "completions/min_length": 6577.0, "completions/min_terminated_length": 6577.0, "epoch": 0.0015130844598784572, "grad_norm": 3.175959825515747, "learning_rate": 9.7e-07, "loss": -0.707, "num_tokens": 327772.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 61 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3205.0, "completions/mean_length": 5698.5, "completions/mean_terminated_length": 3205.0, "completions/min_length": 3205.0, "completions/min_terminated_length": 3205.0, "epoch": 0.0015378891231551531, "grad_norm": 4.70952033996582, "learning_rate": 9.695e-07, "loss": -0.707, "num_tokens": 331971.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 62 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.001562693786431849, "grad_norm": 0.0, "learning_rate": 9.69e-07, "loss": 0.0, "num_tokens": 332841.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 63 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0015874984497085453, "grad_norm": 0.0, "learning_rate": 9.685e-07, "loss": 0.0, "num_tokens": 333733.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 64 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6850.0, "completions/max_terminated_length": 6850.0, "completions/mean_length": 6166.0, "completions/mean_terminated_length": 6166.0, "completions/min_length": 5482.0, "completions/min_terminated_length": 5482.0, "epoch": 0.0016123031129852412, "grad_norm": 0.0, "learning_rate": 9.679999999999999e-07, "loss": 0.0, "num_tokens": 347003.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 65 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7892.0, "completions/max_terminated_length": 7892.0, "completions/mean_length": 6184.0, "completions/mean_terminated_length": 6184.0, "completions/min_length": 4476.0, "completions/min_terminated_length": 4476.0, "epoch": 0.0016371077762619372, "grad_norm": 2.570572853088379, "learning_rate": 9.675e-07, "loss": 0.1953, "num_tokens": 360187.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 66 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7923.0, "completions/mean_length": 8057.5, "completions/mean_terminated_length": 7923.0, "completions/min_length": 7923.0, "completions/min_terminated_length": 7923.0, "epoch": 0.0016619124395386333, "grad_norm": 0.0, "learning_rate": 9.67e-07, "loss": 0.0, "num_tokens": 369156.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 67 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7431.0, "completions/max_terminated_length": 7431.0, "completions/mean_length": 7287.5, "completions/mean_terminated_length": 7287.5, "completions/min_length": 7144.0, "completions/min_terminated_length": 7144.0, "epoch": 0.0016867171028153293, "grad_norm": 0.0, "learning_rate": 9.665e-07, "loss": 0.0, "num_tokens": 384673.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 68 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0017115217660920252, "grad_norm": 0.0, "learning_rate": 9.66e-07, "loss": 0.0, "num_tokens": 385661.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 69 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3853.0, "completions/max_terminated_length": 3853.0, "completions/mean_length": 3633.0, "completions/mean_terminated_length": 3633.0, "completions/min_length": 3413.0, "completions/min_terminated_length": 3413.0, "epoch": 0.0017363264293687214, "grad_norm": 3.7261362075805664, "learning_rate": 9.655e-07, "loss": 0.0428, "num_tokens": 393767.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 70 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0017611310926454174, "grad_norm": 0.0, "learning_rate": 9.649999999999999e-07, "loss": 0.0, "num_tokens": 394673.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 71 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6324.0, "completions/max_terminated_length": 6324.0, "completions/mean_length": 4301.0, "completions/mean_terminated_length": 4301.0, "completions/min_length": 2278.0, "completions/min_terminated_length": 2278.0, "epoch": 0.0017859357559221133, "grad_norm": 2.9709010124206543, "learning_rate": 9.645e-07, "loss": 0.3325, "num_tokens": 404529.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 72 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7644.0, "completions/max_terminated_length": 7644.0, "completions/mean_length": 7015.0, "completions/mean_terminated_length": 7015.0, "completions/min_length": 6386.0, "completions/min_terminated_length": 6386.0, "epoch": 0.0018107404191988095, "grad_norm": 0.0, "learning_rate": 9.64e-07, "loss": 0.0, "num_tokens": 419401.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 73 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7403.0, "completions/mean_length": 7797.5, "completions/mean_terminated_length": 7403.0, "completions/min_length": 7403.0, "completions/min_terminated_length": 7403.0, "epoch": 0.0018355450824755054, "grad_norm": 3.0093350410461426, "learning_rate": 9.635e-07, "loss": -0.707, "num_tokens": 427768.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 74 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3716.0, "completions/mean_length": 5954.0, "completions/mean_terminated_length": 3716.0, "completions/min_length": 3716.0, "completions/min_terminated_length": 3716.0, "epoch": 0.0018603497457522014, "grad_norm": 4.841500282287598, "learning_rate": 9.63e-07, "loss": -0.707, "num_tokens": 432404.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 75 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0018851544090288973, "grad_norm": 0.0, "learning_rate": 9.624999999999999e-07, "loss": 0.0, "num_tokens": 433376.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 76 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0019099590723055935, "grad_norm": 0.0, "learning_rate": 9.619999999999999e-07, "loss": 0.0, "num_tokens": 434532.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 77 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1466.0, "completions/max_terminated_length": 1466.0, "completions/mean_length": 1441.0, "completions/mean_terminated_length": 1441.0, "completions/min_length": 1416.0, "completions/min_terminated_length": 1416.0, "epoch": 0.0019347637355822895, "grad_norm": 5.187351226806641, "learning_rate": 9.615e-07, "loss": -0.0123, "num_tokens": 438322.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 78 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7422.0, "completions/max_terminated_length": 7422.0, "completions/mean_length": 6584.0, "completions/mean_terminated_length": 6584.0, "completions/min_length": 5746.0, "completions/min_terminated_length": 5746.0, "epoch": 0.0019595683988589854, "grad_norm": 0.0, "learning_rate": 9.61e-07, "loss": 0.0, "num_tokens": 452306.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 79 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7629.0, "completions/mean_length": 7910.5, "completions/mean_terminated_length": 7629.0, "completions/min_length": 7629.0, "completions/min_terminated_length": 7629.0, "epoch": 0.0019843730621356814, "grad_norm": 2.825613260269165, "learning_rate": 9.605e-07, "loss": -0.707, "num_tokens": 460781.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 80 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5833.0, "completions/mean_length": 7012.5, "completions/mean_terminated_length": 5833.0, "completions/min_length": 5833.0, "completions/min_terminated_length": 5833.0, "epoch": 0.0020091777254123773, "grad_norm": 0.0, "learning_rate": 9.6e-07, "loss": 0.0, "num_tokens": 467500.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 81 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 562.0, "completions/max_terminated_length": 562.0, "completions/mean_length": 547.5, "completions/mean_terminated_length": 547.5, "completions/min_length": 533.0, "completions/min_terminated_length": 533.0, "epoch": 0.0020339823886890737, "grad_norm": 0.0, "learning_rate": 9.594999999999999e-07, "loss": 0.0, "num_tokens": 469399.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 82 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5891.0, "completions/mean_length": 7041.5, "completions/mean_terminated_length": 5891.0, "completions/min_length": 5891.0, "completions/min_terminated_length": 5891.0, "epoch": 0.0020587870519657697, "grad_norm": 3.186800479888916, "learning_rate": 9.589999999999998e-07, "loss": -0.707, "num_tokens": 476164.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 83 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2801.0, "completions/mean_length": 5496.5, "completions/mean_terminated_length": 2801.0, "completions/min_length": 2801.0, "completions/min_terminated_length": 2801.0, "epoch": 0.0020835917152424656, "grad_norm": 0.0, "learning_rate": 9.585e-07, "loss": 0.0, "num_tokens": 479905.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 84 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3287.0, "completions/max_terminated_length": 3287.0, "completions/mean_length": 2449.5, "completions/mean_terminated_length": 2449.5, "completions/min_length": 1612.0, "completions/min_terminated_length": 1612.0, "epoch": 0.0021083963785191616, "grad_norm": 0.0, "learning_rate": 9.58e-07, "loss": 0.0, "num_tokens": 485642.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 85 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3221.0, "completions/mean_length": 5706.5, "completions/mean_terminated_length": 3221.0, "completions/min_length": 3221.0, "completions/min_terminated_length": 3221.0, "epoch": 0.0021332010417958575, "grad_norm": 5.386900424957275, "learning_rate": 9.575e-07, "loss": -0.707, "num_tokens": 489755.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 86 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0021580057050725535, "grad_norm": 0.0, "learning_rate": 9.57e-07, "loss": 0.0, "num_tokens": 490759.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 87 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1920.0, "completions/max_terminated_length": 1920.0, "completions/mean_length": 1348.0, "completions/mean_terminated_length": 1348.0, "completions/min_length": 776.0, "completions/min_terminated_length": 776.0, "epoch": 0.00218281036834925, "grad_norm": 0.0, "learning_rate": 9.565e-07, "loss": 0.0, "num_tokens": 494265.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 88 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.002207615031625946, "grad_norm": 0.0, "learning_rate": 9.559999999999998e-07, "loss": 0.0, "num_tokens": 495215.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 89 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1339.0, "completions/max_terminated_length": 1339.0, "completions/mean_length": 1126.5, "completions/mean_terminated_length": 1126.5, "completions/min_length": 914.0, "completions/min_terminated_length": 914.0, "epoch": 0.0022324196949026417, "grad_norm": 0.0, "learning_rate": 9.555e-07, "loss": 0.0, "num_tokens": 498388.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 90 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1152.0, "completions/mean_length": 4672.0, "completions/mean_terminated_length": 1152.0, "completions/min_length": 1152.0, "completions/min_terminated_length": 1152.0, "epoch": 0.0022572243581793377, "grad_norm": 6.972071647644043, "learning_rate": 9.55e-07, "loss": -0.707, "num_tokens": 500416.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 91 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4025.0, "completions/max_terminated_length": 4025.0, "completions/mean_length": 2859.0, "completions/mean_terminated_length": 2859.0, "completions/min_length": 1693.0, "completions/min_terminated_length": 1693.0, "epoch": 0.0022820290214560337, "grad_norm": 0.0, "learning_rate": 9.545e-07, "loss": 0.0, "num_tokens": 507034.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 92 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0023068336847327296, "grad_norm": 0.0, "learning_rate": 9.539999999999999e-07, "loss": 0.0, "num_tokens": 507970.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 93 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5449.0, "completions/max_terminated_length": 5449.0, "completions/mean_length": 4332.0, "completions/mean_terminated_length": 4332.0, "completions/min_length": 3215.0, "completions/min_terminated_length": 3215.0, "epoch": 0.0023316383480094256, "grad_norm": 0.0, "learning_rate": 9.535e-07, "loss": 0.0, "num_tokens": 517518.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 94 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7407.0, "completions/max_terminated_length": 7407.0, "completions/mean_length": 7128.5, "completions/mean_terminated_length": 7128.5, "completions/min_length": 6850.0, "completions/min_terminated_length": 6850.0, "epoch": 0.002356443011286122, "grad_norm": 0.0, "learning_rate": 9.529999999999999e-07, "loss": 0.0, "num_tokens": 532715.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 95 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3198.0, "completions/max_terminated_length": 3198.0, "completions/mean_length": 2997.5, "completions/mean_terminated_length": 2997.5, "completions/min_length": 2797.0, "completions/min_terminated_length": 2797.0, "epoch": 0.002381247674562818, "grad_norm": 0.0, "learning_rate": 9.525e-07, "loss": 0.0, "num_tokens": 539546.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 96 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.002406052337839514, "grad_norm": 0.0, "learning_rate": 9.52e-07, "loss": 0.0, "num_tokens": 540494.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 97 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5627.0, "completions/mean_length": 6909.5, "completions/mean_terminated_length": 5627.0, "completions/min_length": 5627.0, "completions/min_terminated_length": 5627.0, "epoch": 0.00243085700111621, "grad_norm": 3.1484227180480957, "learning_rate": 9.515e-07, "loss": -0.707, "num_tokens": 547177.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 98 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1394.0, "completions/max_terminated_length": 1394.0, "completions/mean_length": 1151.0, "completions/mean_terminated_length": 1151.0, "completions/min_length": 908.0, "completions/min_terminated_length": 908.0, "epoch": 0.0024556616643929058, "grad_norm": 0.0, "learning_rate": 9.509999999999999e-07, "loss": 0.0, "num_tokens": 550339.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 99 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0024804663276696017, "grad_norm": 0.0, "learning_rate": 9.504999999999999e-07, "loss": 0.0, "num_tokens": 551373.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7430.0, "completions/max_terminated_length": 7430.0, "completions/mean_length": 7325.0, "completions/mean_terminated_length": 7325.0, "completions/min_length": 7220.0, "completions/min_terminated_length": 7220.0, "epoch": 0.002505270990946298, "grad_norm": 2.1009163856506348, "learning_rate": 9.499999999999999e-07, "loss": 0.0101, "num_tokens": 566851.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 101 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.002530075654222994, "grad_norm": 0.0, "learning_rate": 9.495e-07, "loss": 0.0, "num_tokens": 567685.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 102 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4646.0, "completions/mean_length": 6419.0, "completions/mean_terminated_length": 4646.0, "completions/min_length": 4646.0, "completions/min_terminated_length": 4646.0, "epoch": 0.00255488031749969, "grad_norm": 3.820272445678711, "learning_rate": 9.489999999999999e-07, "loss": 0.707, "num_tokens": 573181.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 103 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.002579684980776386, "grad_norm": 0.0, "learning_rate": 9.485e-07, "loss": 0.0, "num_tokens": 574079.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6435.0, "completions/mean_length": 7313.5, "completions/mean_terminated_length": 6435.0, "completions/min_length": 6435.0, "completions/min_terminated_length": 6435.0, "epoch": 0.002604489644053082, "grad_norm": 0.0, "learning_rate": 9.479999999999999e-07, "loss": 0.0, "num_tokens": 581388.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 105 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.002629294307329778, "grad_norm": 0.0, "learning_rate": 9.474999999999999e-07, "loss": 0.0, "num_tokens": 582282.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 772.0, "completions/max_terminated_length": 772.0, "completions/mean_length": 700.5, "completions/mean_terminated_length": 700.5, "completions/min_length": 629.0, "completions/min_terminated_length": 629.0, "epoch": 0.0026540989706064742, "grad_norm": 0.0, "learning_rate": 9.469999999999999e-07, "loss": 0.0, "num_tokens": 584567.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5406.0, "completions/mean_length": 6799.0, "completions/mean_terminated_length": 5406.0, "completions/min_length": 5406.0, "completions/min_terminated_length": 5406.0, "epoch": 0.00267890363388317, "grad_norm": 3.858245611190796, "learning_rate": 9.465e-07, "loss": -0.707, "num_tokens": 590795.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4226.0, "completions/max_terminated_length": 4226.0, "completions/mean_length": 3925.5, "completions/mean_terminated_length": 3925.5, "completions/min_length": 3625.0, "completions/min_terminated_length": 3625.0, "epoch": 0.002703708297159866, "grad_norm": 0.0, "learning_rate": 9.459999999999999e-07, "loss": 0.0, "num_tokens": 599474.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6626.0, "completions/mean_length": 7409.0, "completions/mean_terminated_length": 6626.0, "completions/min_length": 6626.0, "completions/min_terminated_length": 6626.0, "epoch": 0.002728512960436562, "grad_norm": 2.682093858718872, "learning_rate": 9.455e-07, "loss": -0.707, "num_tokens": 606954.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4949.0, "completions/max_terminated_length": 4949.0, "completions/mean_length": 4065.0, "completions/mean_terminated_length": 4065.0, "completions/min_length": 3181.0, "completions/min_terminated_length": 3181.0, "epoch": 0.002753317623713258, "grad_norm": 0.0, "learning_rate": 9.45e-07, "loss": 0.0, "num_tokens": 615950.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3841.0, "completions/mean_length": 6016.5, "completions/mean_terminated_length": 3841.0, "completions/min_length": 3841.0, "completions/min_terminated_length": 3841.0, "epoch": 0.002778122286989954, "grad_norm": 0.0, "learning_rate": 9.444999999999999e-07, "loss": 0.0, "num_tokens": 620873.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8067.0, "completions/max_terminated_length": 8067.0, "completions/mean_length": 6848.5, "completions/mean_terminated_length": 6848.5, "completions/min_length": 5630.0, "completions/min_terminated_length": 5630.0, "epoch": 0.00280292695026665, "grad_norm": 2.2464535236358643, "learning_rate": 9.439999999999999e-07, "loss": 0.1258, "num_tokens": 635484.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 113 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0028277316135433463, "grad_norm": 0.0, "learning_rate": 9.434999999999999e-07, "loss": 0.0, "num_tokens": 636438.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2228.0, "completions/max_terminated_length": 2228.0, "completions/mean_length": 1777.0, "completions/mean_terminated_length": 1777.0, "completions/min_length": 1326.0, "completions/min_terminated_length": 1326.0, "epoch": 0.0028525362768200423, "grad_norm": 0.0, "learning_rate": 9.429999999999999e-07, "loss": 0.0, "num_tokens": 640838.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 115 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0028773409400967382, "grad_norm": 0.0, "learning_rate": 9.425e-07, "loss": 0.0, "num_tokens": 641686.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4403.0, "completions/mean_length": 6297.5, "completions/mean_terminated_length": 4403.0, "completions/min_length": 4403.0, "completions/min_terminated_length": 4403.0, "epoch": 0.002902145603373434, "grad_norm": 0.0, "learning_rate": 9.419999999999999e-07, "loss": 0.0, "num_tokens": 647239.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 117 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.00292695026665013, "grad_norm": 0.0, "learning_rate": 9.415e-07, "loss": 0.0, "num_tokens": 648203.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 118 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.002951754929926826, "grad_norm": 0.0, "learning_rate": 9.409999999999999e-07, "loss": 0.0, "num_tokens": 649165.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1523.0, "completions/max_terminated_length": 1523.0, "completions/mean_length": 1113.5, "completions/mean_terminated_length": 1113.5, "completions/min_length": 704.0, "completions/min_terminated_length": 704.0, "epoch": 0.0029765595932035225, "grad_norm": 0.0, "learning_rate": 9.404999999999999e-07, "loss": 0.0, "num_tokens": 652186.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5157.0, "completions/mean_length": 6674.5, "completions/mean_terminated_length": 5157.0, "completions/min_length": 5157.0, "completions/min_terminated_length": 5157.0, "epoch": 0.0030013642564802184, "grad_norm": 3.7442312240600586, "learning_rate": 9.399999999999999e-07, "loss": -0.707, "num_tokens": 658269.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4652.0, "completions/max_terminated_length": 4652.0, "completions/mean_length": 4119.5, "completions/mean_terminated_length": 4119.5, "completions/min_length": 3587.0, "completions/min_terminated_length": 3587.0, "epoch": 0.0030261689197569144, "grad_norm": 3.329113483428955, "learning_rate": 9.395e-07, "loss": 0.0914, "num_tokens": 667742.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 635.0, "completions/max_terminated_length": 635.0, "completions/mean_length": 622.5, "completions/mean_terminated_length": 622.5, "completions/min_length": 610.0, "completions/min_terminated_length": 610.0, "epoch": 0.0030509735830336103, "grad_norm": 0.0, "learning_rate": 9.389999999999999e-07, "loss": 0.0, "num_tokens": 669787.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 123 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0030757782463103063, "grad_norm": 0.0, "learning_rate": 9.385e-07, "loss": 0.0, "num_tokens": 670835.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1104.0, "completions/max_terminated_length": 1104.0, "completions/mean_length": 892.5, "completions/mean_terminated_length": 892.5, "completions/min_length": 681.0, "completions/min_terminated_length": 681.0, "epoch": 0.0031005829095870022, "grad_norm": 7.251551151275635, "learning_rate": 9.379999999999998e-07, "loss": -0.1675, "num_tokens": 673424.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 125 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.003125387572863698, "grad_norm": 0.0, "learning_rate": 9.374999999999999e-07, "loss": 0.0, "num_tokens": 674430.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2124.0, "completions/max_terminated_length": 2124.0, "completions/mean_length": 1454.5, "completions/mean_terminated_length": 1454.5, "completions/min_length": 785.0, "completions/min_terminated_length": 785.0, "epoch": 0.0031501922361403946, "grad_norm": 4.2739458084106445, "learning_rate": 9.37e-07, "loss": 0.3254, "num_tokens": 678183.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3147.0, "completions/max_terminated_length": 3147.0, "completions/mean_length": 2157.5, "completions/mean_terminated_length": 2157.5, "completions/min_length": 1168.0, "completions/min_terminated_length": 1168.0, "epoch": 0.0031749968994170905, "grad_norm": 4.0321526527404785, "learning_rate": 9.365e-07, "loss": -0.3243, "num_tokens": 683432.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4937.0, "completions/mean_length": 6564.5, "completions/mean_terminated_length": 4937.0, "completions/min_length": 4937.0, "completions/min_terminated_length": 4937.0, "epoch": 0.0031998015626937865, "grad_norm": 4.583209037780762, "learning_rate": 9.36e-07, "loss": 0.707, "num_tokens": 689271.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3794.0, "completions/max_terminated_length": 3794.0, "completions/mean_length": 2967.5, "completions/mean_terminated_length": 2967.5, "completions/min_length": 2141.0, "completions/min_terminated_length": 2141.0, "epoch": 0.0032246062259704824, "grad_norm": 0.0, "learning_rate": 9.355e-07, "loss": 0.0, "num_tokens": 696246.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2158.0, "completions/max_terminated_length": 2158.0, "completions/mean_length": 1447.0, "completions/mean_terminated_length": 1447.0, "completions/min_length": 736.0, "completions/min_terminated_length": 736.0, "epoch": 0.0032494108892471784, "grad_norm": 5.252513408660889, "learning_rate": 9.35e-07, "loss": 0.3474, "num_tokens": 700006.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 131 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0032742155525238743, "grad_norm": 0.0, "learning_rate": 9.344999999999999e-07, "loss": 0.0, "num_tokens": 701276.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6763.0, "completions/max_terminated_length": 6763.0, "completions/mean_length": 5210.0, "completions/mean_terminated_length": 5210.0, "completions/min_length": 3657.0, "completions/min_terminated_length": 3657.0, "epoch": 0.0032990202158005707, "grad_norm": 0.0, "learning_rate": 9.34e-07, "loss": 0.0, "num_tokens": 712548.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2228.0, "completions/max_terminated_length": 2228.0, "completions/mean_length": 1838.0, "completions/mean_terminated_length": 1838.0, "completions/min_length": 1448.0, "completions/min_terminated_length": 1448.0, "epoch": 0.0033238248790772667, "grad_norm": 0.0, "learning_rate": 9.334999999999999e-07, "loss": 0.0, "num_tokens": 717060.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7443.0, "completions/max_terminated_length": 7443.0, "completions/mean_length": 5863.0, "completions/mean_terminated_length": 5863.0, "completions/min_length": 4283.0, "completions/min_terminated_length": 4283.0, "epoch": 0.0033486295423539626, "grad_norm": 0.0, "learning_rate": 9.33e-07, "loss": 0.0, "num_tokens": 729738.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7805.0, "completions/mean_length": 7998.5, "completions/mean_terminated_length": 7805.0, "completions/min_length": 7805.0, "completions/min_terminated_length": 7805.0, "epoch": 0.0033734342056306586, "grad_norm": 3.191218852996826, "learning_rate": 9.325e-07, "loss": -0.707, "num_tokens": 738385.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 136 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0033982388689073545, "grad_norm": 0.0, "learning_rate": 9.32e-07, "loss": 0.0, "num_tokens": 739291.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2086.0, "completions/max_terminated_length": 2086.0, "completions/mean_length": 1630.0, "completions/mean_terminated_length": 1630.0, "completions/min_length": 1174.0, "completions/min_terminated_length": 1174.0, "epoch": 0.0034230435321840505, "grad_norm": 0.0, "learning_rate": 9.315e-07, "loss": 0.0, "num_tokens": 743413.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6188.0, "completions/mean_length": 7190.0, "completions/mean_terminated_length": 6188.0, "completions/min_length": 6188.0, "completions/min_terminated_length": 6188.0, "epoch": 0.0034478481954607464, "grad_norm": 0.0, "learning_rate": 9.31e-07, "loss": 0.0, "num_tokens": 750477.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2105.0, "completions/max_terminated_length": 2105.0, "completions/mean_length": 2078.5, "completions/mean_terminated_length": 2078.5, "completions/min_length": 2052.0, "completions/min_terminated_length": 2052.0, "epoch": 0.003472652858737443, "grad_norm": 0.0, "learning_rate": 9.304999999999999e-07, "loss": 0.0, "num_tokens": 755630.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 856.0, "completions/max_terminated_length": 856.0, "completions/mean_length": 794.0, "completions/mean_terminated_length": 794.0, "completions/min_length": 732.0, "completions/min_terminated_length": 732.0, "epoch": 0.0034974575220141388, "grad_norm": 0.0, "learning_rate": 9.3e-07, "loss": 0.0, "num_tokens": 758110.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 141 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0035222621852908347, "grad_norm": 0.0, "learning_rate": 9.295e-07, "loss": 0.0, "num_tokens": 759032.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6475.0, "completions/max_terminated_length": 6475.0, "completions/mean_length": 4831.5, "completions/mean_terminated_length": 4831.5, "completions/min_length": 3188.0, "completions/min_terminated_length": 3188.0, "epoch": 0.0035470668485675307, "grad_norm": 0.0, "learning_rate": 9.29e-07, "loss": 0.0, "num_tokens": 769571.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4654.0, "completions/max_terminated_length": 4654.0, "completions/mean_length": 3471.0, "completions/mean_terminated_length": 3471.0, "completions/min_length": 2288.0, "completions/min_terminated_length": 2288.0, "epoch": 0.0035718715118442266, "grad_norm": 0.0, "learning_rate": 9.285e-07, "loss": 0.0, "num_tokens": 777437.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2953.0, "completions/mean_length": 5572.5, "completions/mean_terminated_length": 2953.0, "completions/min_length": 2953.0, "completions/min_terminated_length": 2953.0, "epoch": 0.0035966761751209226, "grad_norm": 4.740444183349609, "learning_rate": 9.28e-07, "loss": -0.707, "num_tokens": 781254.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2241.0, "completions/max_terminated_length": 2241.0, "completions/mean_length": 2079.5, "completions/mean_terminated_length": 2079.5, "completions/min_length": 1918.0, "completions/min_terminated_length": 1918.0, "epoch": 0.003621480838397619, "grad_norm": 0.0, "learning_rate": 9.274999999999999e-07, "loss": 0.0, "num_tokens": 786363.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3387.0, "completions/mean_length": 5789.5, "completions/mean_terminated_length": 3387.0, "completions/min_length": 3387.0, "completions/min_terminated_length": 3387.0, "epoch": 0.003646285501674315, "grad_norm": 3.8157169818878174, "learning_rate": 9.27e-07, "loss": -0.707, "num_tokens": 790712.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2751.0, "completions/mean_length": 5471.5, "completions/mean_terminated_length": 2751.0, "completions/min_length": 2751.0, "completions/min_terminated_length": 2751.0, "epoch": 0.003671090164951011, "grad_norm": 4.982819080352783, "learning_rate": 9.264999999999999e-07, "loss": -0.707, "num_tokens": 794371.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1328.0, "completions/max_terminated_length": 1328.0, "completions/mean_length": 910.5, "completions/mean_terminated_length": 910.5, "completions/min_length": 493.0, "completions/min_terminated_length": 493.0, "epoch": 0.003695894828227707, "grad_norm": 0.0, "learning_rate": 9.26e-07, "loss": 0.0, "num_tokens": 797032.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4380.0, "completions/max_terminated_length": 4380.0, "completions/mean_length": 4266.5, "completions/mean_terminated_length": 4266.5, "completions/min_length": 4153.0, "completions/min_terminated_length": 4153.0, "epoch": 0.0037206994915044028, "grad_norm": 3.322387933731079, "learning_rate": 9.255e-07, "loss": 0.0188, "num_tokens": 806501.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 150 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0037455041547810987, "grad_norm": 0.0, "learning_rate": 9.25e-07, "loss": 0.0, "num_tokens": 807363.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7584.0, "completions/max_terminated_length": 7584.0, "completions/mean_length": 7015.5, "completions/mean_terminated_length": 7015.5, "completions/min_length": 6447.0, "completions/min_terminated_length": 6447.0, "epoch": 0.0037703088180577947, "grad_norm": 0.0, "learning_rate": 9.244999999999999e-07, "loss": 0.0, "num_tokens": 822314.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3456.0, "completions/max_terminated_length": 3456.0, "completions/mean_length": 3390.0, "completions/mean_terminated_length": 3390.0, "completions/min_length": 3324.0, "completions/min_terminated_length": 3324.0, "epoch": 0.003795113481334491, "grad_norm": 0.0, "learning_rate": 9.24e-07, "loss": 0.0, "num_tokens": 830152.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 520.0, "completions/max_terminated_length": 520.0, "completions/mean_length": 498.5, "completions/mean_terminated_length": 498.5, "completions/min_length": 477.0, "completions/min_terminated_length": 477.0, "epoch": 0.003819918144611187, "grad_norm": 0.0, "learning_rate": 9.234999999999999e-07, "loss": 0.0, "num_tokens": 832017.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 154 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.003844722807887883, "grad_norm": 0.0, "learning_rate": 9.23e-07, "loss": 0.0, "num_tokens": 832919.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7425.0, "completions/max_terminated_length": 7425.0, "completions/mean_length": 6627.5, "completions/mean_terminated_length": 6627.5, "completions/min_length": 5830.0, "completions/min_terminated_length": 5830.0, "epoch": 0.003869527471164579, "grad_norm": 1.8814997673034668, "learning_rate": 9.225e-07, "loss": 0.0851, "num_tokens": 847090.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4930.0, "completions/max_terminated_length": 4930.0, "completions/mean_length": 3128.0, "completions/mean_terminated_length": 3128.0, "completions/min_length": 1326.0, "completions/min_terminated_length": 1326.0, "epoch": 0.003894332134441275, "grad_norm": 0.0, "learning_rate": 9.22e-07, "loss": 0.0, "num_tokens": 854212.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7943.0, "completions/mean_length": 8067.5, "completions/mean_terminated_length": 7943.0, "completions/min_length": 7943.0, "completions/min_terminated_length": 7943.0, "epoch": 0.003919136797717971, "grad_norm": 3.1493873596191406, "learning_rate": 9.215e-07, "loss": -0.707, "num_tokens": 862987.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 158 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.003943941460994667, "grad_norm": 0.0, "learning_rate": 9.21e-07, "loss": 0.0, "num_tokens": 863847.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5039.0, "completions/max_terminated_length": 5039.0, "completions/mean_length": 5006.5, "completions/mean_terminated_length": 5006.5, "completions/min_length": 4974.0, "completions/min_terminated_length": 4974.0, "epoch": 0.003968746124271363, "grad_norm": 0.0, "learning_rate": 9.204999999999999e-07, "loss": 0.0, "num_tokens": 874988.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3524.0, "completions/max_terminated_length": 3524.0, "completions/mean_length": 3039.5, "completions/mean_terminated_length": 3039.5, "completions/min_length": 2555.0, "completions/min_terminated_length": 2555.0, "epoch": 0.003993550787548059, "grad_norm": 2.781982183456421, "learning_rate": 9.2e-07, "loss": -0.1127, "num_tokens": 882145.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 161 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004018355450824755, "grad_norm": 0.0, "learning_rate": 9.194999999999999e-07, "loss": 0.0, "num_tokens": 883167.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6782.0, "completions/mean_length": 7487.0, "completions/mean_terminated_length": 6782.0, "completions/min_length": 6782.0, "completions/min_terminated_length": 6782.0, "epoch": 0.004043160114101451, "grad_norm": 0.0, "learning_rate": 9.19e-07, "loss": 0.0, "num_tokens": 891009.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1559.0, "completions/mean_length": 4875.5, "completions/mean_terminated_length": 1559.0, "completions/min_length": 1559.0, "completions/min_terminated_length": 1559.0, "epoch": 0.004067964777378147, "grad_norm": 0.0, "learning_rate": 9.185e-07, "loss": 0.0, "num_tokens": 893514.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6224.0, "completions/mean_length": 7208.0, "completions/mean_terminated_length": 6224.0, "completions/min_length": 6224.0, "completions/min_terminated_length": 6224.0, "epoch": 0.004092769440654843, "grad_norm": 3.8740170001983643, "learning_rate": 9.18e-07, "loss": -0.707, "num_tokens": 900596.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2898.0, "completions/max_terminated_length": 2898.0, "completions/mean_length": 2819.5, "completions/mean_terminated_length": 2819.5, "completions/min_length": 2741.0, "completions/min_terminated_length": 2741.0, "epoch": 0.004117574103931539, "grad_norm": 0.0, "learning_rate": 9.174999999999999e-07, "loss": 0.0, "num_tokens": 907083.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2154.0, "completions/max_terminated_length": 2154.0, "completions/mean_length": 1967.0, "completions/mean_terminated_length": 1967.0, "completions/min_length": 1780.0, "completions/min_terminated_length": 1780.0, "epoch": 0.004142378767208235, "grad_norm": 0.0, "learning_rate": 9.17e-07, "loss": 0.0, "num_tokens": 911919.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4100.0, "completions/max_terminated_length": 4100.0, "completions/mean_length": 3234.0, "completions/mean_terminated_length": 3234.0, "completions/min_length": 2368.0, "completions/min_terminated_length": 2368.0, "epoch": 0.004167183430484931, "grad_norm": 3.112361431121826, "learning_rate": 9.164999999999999e-07, "loss": -0.1893, "num_tokens": 919667.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4527.0, "completions/mean_length": 6359.5, "completions/mean_terminated_length": 4527.0, "completions/min_length": 4527.0, "completions/min_terminated_length": 4527.0, "epoch": 0.004191988093761628, "grad_norm": 3.864109516143799, "learning_rate": 9.16e-07, "loss": -0.707, "num_tokens": 925170.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 169 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004216792757038323, "grad_norm": 0.0, "learning_rate": 9.155e-07, "loss": 0.0, "num_tokens": 926064.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 170 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0042415974203150195, "grad_norm": 0.0, "learning_rate": 9.15e-07, "loss": 0.0, "num_tokens": 926972.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6460.0, "completions/mean_length": 7326.0, "completions/mean_terminated_length": 6460.0, "completions/min_length": 6460.0, "completions/min_terminated_length": 6460.0, "epoch": 0.004266402083591715, "grad_norm": 0.0, "learning_rate": 9.145e-07, "loss": 0.0, "num_tokens": 934332.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 172 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6634.0, "completions/mean_length": 7413.0, "completions/mean_terminated_length": 6634.0, "completions/min_length": 6634.0, "completions/min_terminated_length": 6634.0, "epoch": 0.004291206746868411, "grad_norm": 0.0, "learning_rate": 9.14e-07, "loss": 0.0, "num_tokens": 941854.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4176.0, "completions/max_terminated_length": 4176.0, "completions/mean_length": 3501.0, "completions/mean_terminated_length": 3501.0, "completions/min_length": 2826.0, "completions/min_terminated_length": 2826.0, "epoch": 0.004316011410145107, "grad_norm": 2.747648239135742, "learning_rate": 9.134999999999999e-07, "loss": -0.1363, "num_tokens": 949678.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 174 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004340816073421803, "grad_norm": 0.0, "learning_rate": 9.13e-07, "loss": 0.0, "num_tokens": 950680.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 175 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0043656207366985, "grad_norm": 0.0, "learning_rate": 9.124999999999999e-07, "loss": 0.0, "num_tokens": 951668.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1086.0, "completions/max_terminated_length": 1086.0, "completions/mean_length": 880.0, "completions/mean_terminated_length": 880.0, "completions/min_length": 674.0, "completions/min_terminated_length": 674.0, "epoch": 0.004390425399975195, "grad_norm": 6.485786437988281, "learning_rate": 9.12e-07, "loss": -0.1655, "num_tokens": 954262.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7763.0, "completions/max_terminated_length": 7763.0, "completions/mean_length": 5420.5, "completions/mean_terminated_length": 5420.5, "completions/min_length": 3078.0, "completions/min_terminated_length": 3078.0, "epoch": 0.004415230063251892, "grad_norm": 2.5428731441497803, "learning_rate": 9.115e-07, "loss": -0.3055, "num_tokens": 965931.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4893.0, "completions/mean_length": 6542.5, "completions/mean_terminated_length": 4893.0, "completions/min_length": 4893.0, "completions/min_terminated_length": 4893.0, "epoch": 0.004440034726528587, "grad_norm": 5.310830116271973, "learning_rate": 9.109999999999999e-07, "loss": -0.707, "num_tokens": 971682.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5452.0, "completions/mean_length": 6822.0, "completions/mean_terminated_length": 5452.0, "completions/min_length": 5452.0, "completions/min_terminated_length": 5452.0, "epoch": 0.0044648393898052835, "grad_norm": 4.14500093460083, "learning_rate": 9.104999999999999e-07, "loss": -0.707, "num_tokens": 978092.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 180 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004489644053081979, "grad_norm": 0.0, "learning_rate": 9.1e-07, "loss": 0.0, "num_tokens": 978970.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8051.0, "completions/max_terminated_length": 8051.0, "completions/mean_length": 7415.5, "completions/mean_terminated_length": 7415.5, "completions/min_length": 6780.0, "completions/min_terminated_length": 6780.0, "epoch": 0.004514448716358675, "grad_norm": 0.0, "learning_rate": 9.094999999999999e-07, "loss": 0.0, "num_tokens": 994675.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 182 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004539253379635372, "grad_norm": 0.0, "learning_rate": 9.09e-07, "loss": 0.0, "num_tokens": 995869.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 183 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004564058042912067, "grad_norm": 0.0, "learning_rate": 9.085e-07, "loss": 0.0, "num_tokens": 996765.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8156.0, "completions/max_terminated_length": 8156.0, "completions/mean_length": 6757.0, "completions/mean_terminated_length": 6757.0, "completions/min_length": 5358.0, "completions/min_terminated_length": 5358.0, "epoch": 0.004588862706188764, "grad_norm": 2.1680071353912354, "learning_rate": 9.08e-07, "loss": 0.1464, "num_tokens": 1011095.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 185 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004613667369465459, "grad_norm": 0.0, "learning_rate": 9.074999999999999e-07, "loss": 0.0, "num_tokens": 1012077.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3436.0, "completions/mean_length": 5814.0, "completions/mean_terminated_length": 3436.0, "completions/min_length": 3436.0, "completions/min_terminated_length": 3436.0, "epoch": 0.004638472032742156, "grad_norm": 4.485501289367676, "learning_rate": 9.07e-07, "loss": -0.707, "num_tokens": 1016359.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1739.0, "completions/max_terminated_length": 1739.0, "completions/mean_length": 1568.0, "completions/mean_terminated_length": 1568.0, "completions/min_length": 1397.0, "completions/min_terminated_length": 1397.0, "epoch": 0.004663276696018851, "grad_norm": 0.0, "learning_rate": 9.064999999999999e-07, "loss": 0.0, "num_tokens": 1020389.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2860.0, "completions/max_terminated_length": 2860.0, "completions/mean_length": 2491.0, "completions/mean_terminated_length": 2491.0, "completions/min_length": 2122.0, "completions/min_terminated_length": 2122.0, "epoch": 0.0046880813592955475, "grad_norm": 3.4866182804107666, "learning_rate": 9.06e-07, "loss": 0.1047, "num_tokens": 1026269.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2420.0, "completions/max_terminated_length": 2420.0, "completions/mean_length": 2315.5, "completions/mean_terminated_length": 2315.5, "completions/min_length": 2211.0, "completions/min_terminated_length": 2211.0, "epoch": 0.004712886022572244, "grad_norm": 0.0, "learning_rate": 9.055e-07, "loss": 0.0, "num_tokens": 1031780.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 190 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004737690685848939, "grad_norm": 0.0, "learning_rate": 9.05e-07, "loss": 0.0, "num_tokens": 1032732.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 191 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004762495349125636, "grad_norm": 0.0, "learning_rate": 9.045e-07, "loss": 0.0, "num_tokens": 1033662.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 192 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.004787300012402331, "grad_norm": 0.0, "learning_rate": 9.039999999999999e-07, "loss": 0.0, "num_tokens": 1034600.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 958.0, "completions/max_terminated_length": 958.0, "completions/mean_length": 834.5, "completions/mean_terminated_length": 834.5, "completions/min_length": 711.0, "completions/min_terminated_length": 711.0, "epoch": 0.004812104675679028, "grad_norm": 0.0, "learning_rate": 9.034999999999999e-07, "loss": 0.0, "num_tokens": 1037103.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4846.0, "completions/max_terminated_length": 4846.0, "completions/mean_length": 4107.5, "completions/mean_terminated_length": 4107.5, "completions/min_length": 3369.0, "completions/min_terminated_length": 3369.0, "epoch": 0.004836909338955724, "grad_norm": 0.0, "learning_rate": 9.03e-07, "loss": 0.0, "num_tokens": 1046132.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 868.0, "completions/max_terminated_length": 868.0, "completions/mean_length": 785.0, "completions/mean_terminated_length": 785.0, "completions/min_length": 702.0, "completions/min_terminated_length": 702.0, "epoch": 0.00486171400223242, "grad_norm": 0.0, "learning_rate": 9.024999999999999e-07, "loss": 0.0, "num_tokens": 1048578.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5038.0, "completions/max_terminated_length": 5038.0, "completions/mean_length": 3663.5, "completions/mean_terminated_length": 3663.5, "completions/min_length": 2289.0, "completions/min_terminated_length": 2289.0, "epoch": 0.004886518665509116, "grad_norm": 0.0, "learning_rate": 9.02e-07, "loss": 0.0, "num_tokens": 1056803.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 197 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0049113233287858115, "grad_norm": 0.0, "learning_rate": 9.015e-07, "loss": 0.0, "num_tokens": 1057705.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2718.0, "completions/max_terminated_length": 2718.0, "completions/mean_length": 1657.0, "completions/mean_terminated_length": 1657.0, "completions/min_length": 596.0, "completions/min_terminated_length": 596.0, "epoch": 0.004936127992062508, "grad_norm": 4.324167251586914, "learning_rate": 9.01e-07, "loss": -0.4527, "num_tokens": 1061807.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6075.0, "completions/mean_length": 7133.5, "completions/mean_terminated_length": 6075.0, "completions/min_length": 6075.0, "completions/min_terminated_length": 6075.0, "epoch": 0.004960932655339203, "grad_norm": 3.7080907821655273, "learning_rate": 9.004999999999999e-07, "loss": -0.707, "num_tokens": 1068746.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1725.0, "completions/max_terminated_length": 1725.0, "completions/mean_length": 1526.5, "completions/mean_terminated_length": 1526.5, "completions/min_length": 1328.0, "completions/min_terminated_length": 1328.0, "epoch": 0.0049857373186159, "grad_norm": 0.0, "learning_rate": 9e-07, "loss": 0.0, "num_tokens": 1072673.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2349.0, "completions/mean_length": 5270.5, "completions/mean_terminated_length": 2349.0, "completions/min_length": 2349.0, "completions/min_terminated_length": 2349.0, "epoch": 0.005010541981892596, "grad_norm": 5.605862140655518, "learning_rate": 8.994999999999999e-07, "loss": -0.707, "num_tokens": 1075876.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 202 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8191.0, "completions/max_terminated_length": 8191.0, "completions/mean_length": 5560.5, "completions/mean_terminated_length": 5560.5, "completions/min_length": 2930.0, "completions/min_terminated_length": 2930.0, "epoch": 0.005035346645169292, "grad_norm": 0.0, "learning_rate": 8.99e-07, "loss": 0.0, "num_tokens": 1087821.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 614.0, "completions/max_terminated_length": 614.0, "completions/mean_length": 613.5, "completions/mean_terminated_length": 613.5, "completions/min_length": 613.0, "completions/min_terminated_length": 613.0, "epoch": 0.005060151308445988, "grad_norm": 8.131155014038086, "learning_rate": 8.985e-07, "loss": -0.0006, "num_tokens": 1089842.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 204 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005084955971722684, "grad_norm": 0.0, "learning_rate": 8.98e-07, "loss": 0.0, "num_tokens": 1090678.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4937.0, "completions/max_terminated_length": 4937.0, "completions/mean_length": 3986.0, "completions/mean_terminated_length": 3986.0, "completions/min_length": 3035.0, "completions/min_terminated_length": 3035.0, "epoch": 0.00510976063499938, "grad_norm": 0.0, "learning_rate": 8.974999999999999e-07, "loss": 0.0, "num_tokens": 1099560.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7967.0, "completions/max_terminated_length": 7967.0, "completions/mean_length": 7526.0, "completions/mean_terminated_length": 7526.0, "completions/min_length": 7085.0, "completions/min_terminated_length": 7085.0, "epoch": 0.0051345652982760755, "grad_norm": 0.0, "learning_rate": 8.969999999999999e-07, "loss": 0.0, "num_tokens": 1115454.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 207 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005159369961552772, "grad_norm": 0.0, "learning_rate": 8.964999999999999e-07, "loss": 0.0, "num_tokens": 1116494.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 208 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005184174624829468, "grad_norm": 0.0, "learning_rate": 8.96e-07, "loss": 0.0, "num_tokens": 1117406.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 209 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005208979288106164, "grad_norm": 0.0, "learning_rate": 8.954999999999999e-07, "loss": 0.0, "num_tokens": 1118316.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8172.0, "completions/max_terminated_length": 8172.0, "completions/mean_length": 7076.5, "completions/mean_terminated_length": 7076.5, "completions/min_length": 5981.0, "completions/min_terminated_length": 5981.0, "epoch": 0.00523378395138286, "grad_norm": 0.0, "learning_rate": 8.95e-07, "loss": 0.0, "num_tokens": 1133383.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 211 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5058.0, "completions/max_terminated_length": 5058.0, "completions/mean_length": 4607.0, "completions/mean_terminated_length": 4607.0, "completions/min_length": 4156.0, "completions/min_terminated_length": 4156.0, "epoch": 0.005258588614659556, "grad_norm": 2.3106257915496826, "learning_rate": 8.945e-07, "loss": 0.0692, "num_tokens": 1143415.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 212 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005283393277936252, "grad_norm": 0.0, "learning_rate": 8.939999999999999e-07, "loss": 0.0, "num_tokens": 1144461.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3730.0, "completions/mean_length": 5961.0, "completions/mean_terminated_length": 3730.0, "completions/min_length": 3730.0, "completions/min_terminated_length": 3730.0, "epoch": 0.0053081979412129485, "grad_norm": 4.364687442779541, "learning_rate": 8.934999999999999e-07, "loss": -0.707, "num_tokens": 1149715.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 214 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005333002604489644, "grad_norm": 0.0, "learning_rate": 8.93e-07, "loss": 0.0, "num_tokens": 1150687.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6863.0, "completions/mean_length": 7527.5, "completions/mean_terminated_length": 6863.0, "completions/min_length": 6863.0, "completions/min_terminated_length": 6863.0, "epoch": 0.00535780726776634, "grad_norm": 0.0, "learning_rate": 8.924999999999999e-07, "loss": 0.0, "num_tokens": 1158562.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7927.0, "completions/mean_length": 8059.5, "completions/mean_terminated_length": 7927.0, "completions/min_length": 7927.0, "completions/min_terminated_length": 7927.0, "epoch": 0.005382611931043036, "grad_norm": 0.0, "learning_rate": 8.92e-07, "loss": 0.0, "num_tokens": 1167347.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 217 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8068.0, "completions/max_terminated_length": 8068.0, "completions/mean_length": 7102.5, "completions/mean_terminated_length": 7102.5, "completions/min_length": 6137.0, "completions/min_terminated_length": 6137.0, "epoch": 0.005407416594319732, "grad_norm": 0.0, "learning_rate": 8.915e-07, "loss": 0.0, "num_tokens": 1182390.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 218 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005432221257596428, "grad_norm": 0.0, "learning_rate": 8.91e-07, "loss": 0.0, "num_tokens": 1183348.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5538.0, "completions/max_terminated_length": 5538.0, "completions/mean_length": 5052.0, "completions/mean_terminated_length": 5052.0, "completions/min_length": 4566.0, "completions/min_terminated_length": 4566.0, "epoch": 0.005457025920873124, "grad_norm": 2.620621681213379, "learning_rate": 8.904999999999999e-07, "loss": -0.068, "num_tokens": 1194306.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 220 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0054818305841498206, "grad_norm": 0.0, "learning_rate": 8.9e-07, "loss": 0.0, "num_tokens": 1195160.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1261.0, "completions/max_terminated_length": 1261.0, "completions/mean_length": 1174.0, "completions/mean_terminated_length": 1174.0, "completions/min_length": 1087.0, "completions/min_terminated_length": 1087.0, "epoch": 0.005506635247426516, "grad_norm": 0.0, "learning_rate": 8.894999999999999e-07, "loss": 0.0, "num_tokens": 1198310.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 531.0, "completions/max_terminated_length": 531.0, "completions/mean_length": 402.0, "completions/mean_terminated_length": 402.0, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "epoch": 0.0055314399107032125, "grad_norm": 10.103218078613281, "learning_rate": 8.89e-07, "loss": -0.2269, "num_tokens": 1199908.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5418.0, "completions/max_terminated_length": 5418.0, "completions/mean_length": 5034.5, "completions/mean_terminated_length": 5034.5, "completions/min_length": 4651.0, "completions/min_terminated_length": 4651.0, "epoch": 0.005556244573979908, "grad_norm": 0.0, "learning_rate": 8.884999999999999e-07, "loss": 0.0, "num_tokens": 1210915.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1401.0, "completions/max_terminated_length": 1401.0, "completions/mean_length": 1218.5, "completions/mean_terminated_length": 1218.5, "completions/min_length": 1036.0, "completions/min_terminated_length": 1036.0, "epoch": 0.005581049237256604, "grad_norm": 0.0, "learning_rate": 8.88e-07, "loss": 0.0, "num_tokens": 1214190.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 225 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0056058539005333, "grad_norm": 0.0, "learning_rate": 8.874999999999999e-07, "loss": 0.0, "num_tokens": 1215182.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5637.0, "completions/mean_length": 6914.5, "completions/mean_terminated_length": 5637.0, "completions/min_length": 5637.0, "completions/min_terminated_length": 5637.0, "epoch": 0.005630658563809996, "grad_norm": 0.0, "learning_rate": 8.869999999999999e-07, "loss": 0.0, "num_tokens": 1221723.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 227 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005655463227086693, "grad_norm": 0.0, "learning_rate": 8.864999999999999e-07, "loss": 0.0, "num_tokens": 1222777.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1668.0, "completions/max_terminated_length": 1668.0, "completions/mean_length": 1596.5, "completions/mean_terminated_length": 1596.5, "completions/min_length": 1525.0, "completions/min_terminated_length": 1525.0, "epoch": 0.005680267890363388, "grad_norm": 0.0, "learning_rate": 8.86e-07, "loss": 0.0, "num_tokens": 1226778.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 229 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 958.0, "completions/max_terminated_length": 958.0, "completions/mean_length": 892.0, "completions/mean_terminated_length": 892.0, "completions/min_length": 826.0, "completions/min_terminated_length": 826.0, "epoch": 0.005705072553640085, "grad_norm": 0.0, "learning_rate": 8.854999999999999e-07, "loss": 0.0, "num_tokens": 1229432.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7145.0, "completions/max_terminated_length": 7145.0, "completions/mean_length": 5668.0, "completions/mean_terminated_length": 5668.0, "completions/min_length": 4191.0, "completions/min_terminated_length": 4191.0, "epoch": 0.00572987721691678, "grad_norm": 0.0, "learning_rate": 8.85e-07, "loss": 0.0, "num_tokens": 1241916.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 231 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0057546818801934765, "grad_norm": 0.0, "learning_rate": 8.845e-07, "loss": 0.0, "num_tokens": 1242948.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 232 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6420.0, "completions/max_terminated_length": 6420.0, "completions/mean_length": 4314.0, "completions/mean_terminated_length": 4314.0, "completions/min_length": 2208.0, "completions/min_terminated_length": 2208.0, "epoch": 0.005779486543470172, "grad_norm": 0.0, "learning_rate": 8.839999999999999e-07, "loss": 0.0, "num_tokens": 1252572.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 233 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005804291206746868, "grad_norm": 0.0, "learning_rate": 8.834999999999999e-07, "loss": 0.0, "num_tokens": 1253494.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3630.0, "completions/max_terminated_length": 3630.0, "completions/mean_length": 2824.0, "completions/mean_terminated_length": 2824.0, "completions/min_length": 2018.0, "completions/min_terminated_length": 2018.0, "epoch": 0.005829095870023565, "grad_norm": 0.0, "learning_rate": 8.83e-07, "loss": 0.0, "num_tokens": 1260060.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8022.0, "completions/mean_length": 8107.0, "completions/mean_terminated_length": 8022.0, "completions/min_length": 8022.0, "completions/min_terminated_length": 8022.0, "epoch": 0.00585390053330026, "grad_norm": 0.0, "learning_rate": 8.824999999999999e-07, "loss": 0.0, "num_tokens": 1269358.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1506.0, "completions/max_terminated_length": 1506.0, "completions/mean_length": 1290.0, "completions/mean_terminated_length": 1290.0, "completions/min_length": 1074.0, "completions/min_terminated_length": 1074.0, "epoch": 0.005878705196576957, "grad_norm": 0.0, "learning_rate": 8.82e-07, "loss": 0.0, "num_tokens": 1272852.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6084.0, "completions/mean_length": 7138.0, "completions/mean_terminated_length": 6084.0, "completions/min_length": 6084.0, "completions/min_terminated_length": 6084.0, "epoch": 0.005903509859853652, "grad_norm": 3.295689344406128, "learning_rate": 8.814999999999999e-07, "loss": -0.707, "num_tokens": 1279806.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 238 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005928314523130349, "grad_norm": 0.0, "learning_rate": 8.81e-07, "loss": 0.0, "num_tokens": 1280756.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 239 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.005953119186407045, "grad_norm": 0.0, "learning_rate": 8.804999999999999e-07, "loss": 0.0, "num_tokens": 1281724.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1651.0, "completions/max_terminated_length": 1651.0, "completions/mean_length": 1412.5, "completions/mean_terminated_length": 1412.5, "completions/min_length": 1174.0, "completions/min_terminated_length": 1174.0, "epoch": 0.0059779238496837405, "grad_norm": 0.0, "learning_rate": 8.799999999999999e-07, "loss": 0.0, "num_tokens": 1285385.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1465.0, "completions/max_terminated_length": 1465.0, "completions/mean_length": 1395.0, "completions/mean_terminated_length": 1395.0, "completions/min_length": 1325.0, "completions/min_terminated_length": 1325.0, "epoch": 0.006002728512960437, "grad_norm": 5.507553577423096, "learning_rate": 8.794999999999999e-07, "loss": -0.0355, "num_tokens": 1289349.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3090.0, "completions/max_terminated_length": 3090.0, "completions/mean_length": 2367.5, "completions/mean_terminated_length": 2367.5, "completions/min_length": 1645.0, "completions/min_terminated_length": 1645.0, "epoch": 0.006027533176237132, "grad_norm": 0.0, "learning_rate": 8.79e-07, "loss": 0.0, "num_tokens": 1295074.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 243 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3913.0, "completions/mean_length": 6052.5, "completions/mean_terminated_length": 3913.0, "completions/min_length": 3913.0, "completions/min_terminated_length": 3913.0, "epoch": 0.006052337839513829, "grad_norm": 0.0, "learning_rate": 8.784999999999999e-07, "loss": 0.0, "num_tokens": 1299817.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2779.0, "completions/mean_length": 5485.5, "completions/mean_terminated_length": 2779.0, "completions/min_length": 2779.0, "completions/min_terminated_length": 2779.0, "epoch": 0.006077142502790524, "grad_norm": 5.090529918670654, "learning_rate": 8.78e-07, "loss": -0.707, "num_tokens": 1303952.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 245 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006101947166067221, "grad_norm": 0.0, "learning_rate": 8.774999999999999e-07, "loss": 0.0, "num_tokens": 1304786.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4885.0, "completions/max_terminated_length": 4885.0, "completions/mean_length": 4466.5, "completions/mean_terminated_length": 4466.5, "completions/min_length": 4048.0, "completions/min_terminated_length": 4048.0, "epoch": 0.006126751829343917, "grad_norm": 0.0, "learning_rate": 8.769999999999999e-07, "loss": 0.0, "num_tokens": 1314717.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6399.0, "completions/max_terminated_length": 6399.0, "completions/mean_length": 5965.5, "completions/mean_terminated_length": 5965.5, "completions/min_length": 5532.0, "completions/min_terminated_length": 5532.0, "epoch": 0.006151556492620613, "grad_norm": 0.0, "learning_rate": 8.764999999999999e-07, "loss": 0.0, "num_tokens": 1327578.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4238.0, "completions/max_terminated_length": 4238.0, "completions/mean_length": 3777.0, "completions/mean_terminated_length": 3777.0, "completions/min_length": 3316.0, "completions/min_terminated_length": 3316.0, "epoch": 0.006176361155897309, "grad_norm": 0.0, "learning_rate": 8.76e-07, "loss": 0.0, "num_tokens": 1336004.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 249 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7581.0, "completions/mean_length": 7886.5, "completions/mean_terminated_length": 7581.0, "completions/min_length": 7581.0, "completions/min_terminated_length": 7581.0, "epoch": 0.0062011658191740045, "grad_norm": 0.0, "learning_rate": 8.754999999999999e-07, "loss": 0.0, "num_tokens": 1344435.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4178.0, "completions/max_terminated_length": 4178.0, "completions/mean_length": 3899.5, "completions/mean_terminated_length": 3899.5, "completions/min_length": 3621.0, "completions/min_terminated_length": 3621.0, "epoch": 0.006225970482450701, "grad_norm": 0.0, "learning_rate": 8.75e-07, "loss": 0.0, "num_tokens": 1353056.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5710.0, "completions/max_terminated_length": 5710.0, "completions/mean_length": 3907.0, "completions/mean_terminated_length": 3907.0, "completions/min_length": 2104.0, "completions/min_terminated_length": 2104.0, "epoch": 0.006250775145727396, "grad_norm": 0.0, "learning_rate": 8.745000000000001e-07, "loss": 0.0, "num_tokens": 1361786.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 252 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006275579809004093, "grad_norm": 0.0, "learning_rate": 8.739999999999999e-07, "loss": 0.0, "num_tokens": 1362648.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 253 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006300384472280789, "grad_norm": 0.0, "learning_rate": 8.735e-07, "loss": 0.0, "num_tokens": 1363476.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3493.0, "completions/max_terminated_length": 3493.0, "completions/mean_length": 3290.5, "completions/mean_terminated_length": 3290.5, "completions/min_length": 3088.0, "completions/min_terminated_length": 3088.0, "epoch": 0.006325189135557485, "grad_norm": 0.0, "learning_rate": 8.729999999999999e-07, "loss": 0.0, "num_tokens": 1370987.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 255 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006349993798834181, "grad_norm": 0.0, "learning_rate": 8.725e-07, "loss": 0.0, "num_tokens": 1371997.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 256 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006374798462110877, "grad_norm": 0.0, "learning_rate": 8.72e-07, "loss": 0.0, "num_tokens": 1372967.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5594.0, "completions/max_terminated_length": 5594.0, "completions/mean_length": 4537.5, "completions/mean_terminated_length": 4537.5, "completions/min_length": 3481.0, "completions/min_terminated_length": 3481.0, "epoch": 0.006399603125387573, "grad_norm": 0.0, "learning_rate": 8.715e-07, "loss": 0.0, "num_tokens": 1383010.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 258 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4214.0, "completions/max_terminated_length": 4214.0, "completions/mean_length": 3946.5, "completions/mean_terminated_length": 3946.5, "completions/min_length": 3679.0, "completions/min_terminated_length": 3679.0, "epoch": 0.0064244077886642685, "grad_norm": 0.0, "learning_rate": 8.71e-07, "loss": 0.0, "num_tokens": 1391725.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3951.0, "completions/mean_length": 6071.5, "completions/mean_terminated_length": 3951.0, "completions/min_length": 3951.0, "completions/min_terminated_length": 3951.0, "epoch": 0.006449212451940965, "grad_norm": 5.363536357879639, "learning_rate": 8.705e-07, "loss": -0.707, "num_tokens": 1396548.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 260 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006474017115217661, "grad_norm": 0.0, "learning_rate": 8.699999999999999e-07, "loss": 0.0, "num_tokens": 1397424.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 261 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2061.0, "completions/max_terminated_length": 2061.0, "completions/mean_length": 2041.0, "completions/mean_terminated_length": 2041.0, "completions/min_length": 2021.0, "completions/min_terminated_length": 2021.0, "epoch": 0.006498821778494357, "grad_norm": 0.0, "learning_rate": 8.695e-07, "loss": 0.0, "num_tokens": 1402450.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 262 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006523626441771053, "grad_norm": 0.0, "learning_rate": 8.69e-07, "loss": 0.0, "num_tokens": 1403378.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 263 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2944.0, "completions/mean_length": 5568.0, "completions/mean_terminated_length": 2944.0, "completions/min_length": 2944.0, "completions/min_terminated_length": 2944.0, "epoch": 0.006548431105047749, "grad_norm": 0.0, "learning_rate": 8.685e-07, "loss": 0.0, "num_tokens": 1407270.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8105.0, "completions/mean_length": 8148.5, "completions/mean_terminated_length": 8105.0, "completions/min_length": 8105.0, "completions/min_terminated_length": 8105.0, "epoch": 0.006573235768324445, "grad_norm": 0.0, "learning_rate": 8.68e-07, "loss": 0.0, "num_tokens": 1416255.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8040.0, "completions/max_terminated_length": 8040.0, "completions/mean_length": 6077.0, "completions/mean_terminated_length": 6077.0, "completions/min_length": 4114.0, "completions/min_terminated_length": 4114.0, "epoch": 0.0065980404316011414, "grad_norm": 0.0, "learning_rate": 8.675000000000001e-07, "loss": 0.0, "num_tokens": 1429363.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 266 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006622845094877837, "grad_norm": 0.0, "learning_rate": 8.669999999999999e-07, "loss": 0.0, "num_tokens": 1430255.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 267 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006647649758154533, "grad_norm": 0.0, "learning_rate": 8.665e-07, "loss": 0.0, "num_tokens": 1431159.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 268 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006672454421431229, "grad_norm": 0.0, "learning_rate": 8.659999999999999e-07, "loss": 0.0, "num_tokens": 1432031.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 269 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7681.0, "completions/max_terminated_length": 7681.0, "completions/mean_length": 6283.0, "completions/mean_terminated_length": 6283.0, "completions/min_length": 4885.0, "completions/min_terminated_length": 4885.0, "epoch": 0.006697259084707925, "grad_norm": 2.0854103565216064, "learning_rate": 8.655e-07, "loss": -0.1573, "num_tokens": 1445427.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 270 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006722063747984621, "grad_norm": 0.0, "learning_rate": 8.65e-07, "loss": 0.0, "num_tokens": 1446319.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 271 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5493.0, "completions/max_terminated_length": 5493.0, "completions/mean_length": 3223.5, "completions/mean_terminated_length": 3223.5, "completions/min_length": 954.0, "completions/min_terminated_length": 954.0, "epoch": 0.006746868411261317, "grad_norm": 0.0, "learning_rate": 8.645e-07, "loss": 0.0, "num_tokens": 1453606.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 272 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0067716730745380135, "grad_norm": 0.0, "learning_rate": 8.639999999999999e-07, "loss": 0.0, "num_tokens": 1454512.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7169.0, "completions/mean_length": 7680.5, "completions/mean_terminated_length": 7169.0, "completions/min_length": 7169.0, "completions/min_terminated_length": 7169.0, "epoch": 0.006796477737814709, "grad_norm": 3.1028215885162354, "learning_rate": 8.635e-07, "loss": -0.707, "num_tokens": 1462691.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3128.0, "completions/max_terminated_length": 3128.0, "completions/mean_length": 2626.5, "completions/mean_terminated_length": 2626.5, "completions/min_length": 2125.0, "completions/min_terminated_length": 2125.0, "epoch": 0.0068212824010914054, "grad_norm": 3.5959722995758057, "learning_rate": 8.629999999999999e-07, "loss": 0.135, "num_tokens": 1468794.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 275 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006846087064368101, "grad_norm": 0.0, "learning_rate": 8.625e-07, "loss": 0.0, "num_tokens": 1469830.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 276 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006870891727644797, "grad_norm": 0.0, "learning_rate": 8.62e-07, "loss": 0.0, "num_tokens": 1470854.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 277 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006895696390921493, "grad_norm": 0.0, "learning_rate": 8.615e-07, "loss": 0.0, "num_tokens": 1471836.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 278 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6826.0, "completions/mean_length": 7509.0, "completions/mean_terminated_length": 6826.0, "completions/min_length": 6826.0, "completions/min_terminated_length": 6826.0, "epoch": 0.006920501054198189, "grad_norm": 0.0, "learning_rate": 8.61e-07, "loss": 0.0, "num_tokens": 1479590.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3311.0, "completions/max_terminated_length": 3311.0, "completions/mean_length": 3198.0, "completions/mean_terminated_length": 3198.0, "completions/min_length": 3085.0, "completions/min_terminated_length": 3085.0, "epoch": 0.006945305717474886, "grad_norm": 0.0, "learning_rate": 8.605e-07, "loss": 0.0, "num_tokens": 1486838.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 280 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.006970110380751581, "grad_norm": 0.0, "learning_rate": 8.599999999999999e-07, "loss": 0.0, "num_tokens": 1487834.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2602.0, "completions/max_terminated_length": 2602.0, "completions/mean_length": 2547.0, "completions/mean_terminated_length": 2547.0, "completions/min_length": 2492.0, "completions/min_terminated_length": 2492.0, "epoch": 0.0069949150440282775, "grad_norm": 3.4948172569274902, "learning_rate": 8.595e-07, "loss": -0.0153, "num_tokens": 1493814.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3845.0, "completions/max_terminated_length": 3845.0, "completions/mean_length": 3340.0, "completions/mean_terminated_length": 3340.0, "completions/min_length": 2835.0, "completions/min_terminated_length": 2835.0, "epoch": 0.007019719707304973, "grad_norm": 0.0, "learning_rate": 8.59e-07, "loss": 0.0, "num_tokens": 1501346.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 283 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0070445243705816694, "grad_norm": 0.0, "learning_rate": 8.585e-07, "loss": 0.0, "num_tokens": 1502240.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6783.0, "completions/max_terminated_length": 6783.0, "completions/mean_length": 6013.0, "completions/mean_terminated_length": 6013.0, "completions/min_length": 5243.0, "completions/min_terminated_length": 5243.0, "epoch": 0.007069329033858365, "grad_norm": 0.0, "learning_rate": 8.58e-07, "loss": 0.0, "num_tokens": 1515080.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5741.0, "completions/max_terminated_length": 5741.0, "completions/mean_length": 4038.0, "completions/mean_terminated_length": 4038.0, "completions/min_length": 2335.0, "completions/min_terminated_length": 2335.0, "epoch": 0.007094133697135061, "grad_norm": 3.680518388748169, "learning_rate": 8.575e-07, "loss": 0.2982, "num_tokens": 1523988.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 286 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007118938360411758, "grad_norm": 0.0, "learning_rate": 8.569999999999999e-07, "loss": 0.0, "num_tokens": 1524936.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3947.0, "completions/max_terminated_length": 3947.0, "completions/mean_length": 3886.5, "completions/mean_terminated_length": 3886.5, "completions/min_length": 3826.0, "completions/min_terminated_length": 3826.0, "epoch": 0.007143743023688453, "grad_norm": 0.0, "learning_rate": 8.565e-07, "loss": 0.0, "num_tokens": 1533571.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1027.0, "completions/max_terminated_length": 1027.0, "completions/mean_length": 925.5, "completions/mean_terminated_length": 925.5, "completions/min_length": 824.0, "completions/min_terminated_length": 824.0, "epoch": 0.00716854768696515, "grad_norm": 0.0, "learning_rate": 8.559999999999999e-07, "loss": 0.0, "num_tokens": 1536284.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1358.0, "completions/max_terminated_length": 1358.0, "completions/mean_length": 1303.5, "completions/mean_terminated_length": 1303.5, "completions/min_length": 1249.0, "completions/min_terminated_length": 1249.0, "epoch": 0.007193352350241845, "grad_norm": 0.0, "learning_rate": 8.555e-07, "loss": 0.0, "num_tokens": 1539801.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6184.0, "completions/max_terminated_length": 6184.0, "completions/mean_length": 5404.0, "completions/mean_terminated_length": 5404.0, "completions/min_length": 4624.0, "completions/min_terminated_length": 4624.0, "epoch": 0.0072181570135185415, "grad_norm": 0.0, "learning_rate": 8.55e-07, "loss": 0.0, "num_tokens": 1551535.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 291 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5847.0, "completions/max_terminated_length": 5847.0, "completions/mean_length": 4799.0, "completions/mean_terminated_length": 4799.0, "completions/min_length": 3751.0, "completions/min_terminated_length": 3751.0, "epoch": 0.007242961676795238, "grad_norm": 0.0, "learning_rate": 8.545e-07, "loss": 0.0, "num_tokens": 1561985.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1153.0, "completions/max_terminated_length": 1153.0, "completions/mean_length": 1053.5, "completions/mean_terminated_length": 1053.5, "completions/min_length": 954.0, "completions/min_terminated_length": 954.0, "epoch": 0.0072677663400719334, "grad_norm": 0.0, "learning_rate": 8.539999999999999e-07, "loss": 0.0, "num_tokens": 1564896.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 293 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.00729257100334863, "grad_norm": 0.0, "learning_rate": 8.535e-07, "loss": 0.0, "num_tokens": 1566052.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4732.0, "completions/max_terminated_length": 4732.0, "completions/mean_length": 3899.5, "completions/mean_terminated_length": 3899.5, "completions/min_length": 3067.0, "completions/min_terminated_length": 3067.0, "epoch": 0.007317375666625325, "grad_norm": 0.0, "learning_rate": 8.529999999999999e-07, "loss": 0.0, "num_tokens": 1574775.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3624.0, "completions/mean_length": 5908.0, "completions/mean_terminated_length": 3624.0, "completions/min_length": 3624.0, "completions/min_terminated_length": 3624.0, "epoch": 0.007342180329902022, "grad_norm": 0.0, "learning_rate": 8.525e-07, "loss": 0.0, "num_tokens": 1579265.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3974.0, "completions/max_terminated_length": 3974.0, "completions/mean_length": 3705.0, "completions/mean_terminated_length": 3705.0, "completions/min_length": 3436.0, "completions/min_terminated_length": 3436.0, "epoch": 0.007366984993178717, "grad_norm": 3.3195745944976807, "learning_rate": 8.52e-07, "loss": 0.0513, "num_tokens": 1587487.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6362.0, "completions/max_terminated_length": 6362.0, "completions/mean_length": 6209.5, "completions/mean_terminated_length": 6209.5, "completions/min_length": 6057.0, "completions/min_terminated_length": 6057.0, "epoch": 0.007391789656455414, "grad_norm": 0.0, "learning_rate": 8.515e-07, "loss": 0.0, "num_tokens": 1600864.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 298 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.00741659431973211, "grad_norm": 0.0, "learning_rate": 8.51e-07, "loss": 0.0, "num_tokens": 1601796.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4598.0, "completions/max_terminated_length": 4598.0, "completions/mean_length": 3761.0, "completions/mean_terminated_length": 3761.0, "completions/min_length": 2924.0, "completions/min_terminated_length": 2924.0, "epoch": 0.0074413989830088055, "grad_norm": 0.0, "learning_rate": 8.504999999999999e-07, "loss": 0.0, "num_tokens": 1610178.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5139.0, "completions/max_terminated_length": 5139.0, "completions/mean_length": 3686.5, "completions/mean_terminated_length": 3686.5, "completions/min_length": 2234.0, "completions/min_terminated_length": 2234.0, "epoch": 0.007466203646285502, "grad_norm": 0.0, "learning_rate": 8.499999999999999e-07, "loss": 0.0, "num_tokens": 1619335.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 301 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 812.0, "completions/max_terminated_length": 812.0, "completions/mean_length": 703.0, "completions/mean_terminated_length": 703.0, "completions/min_length": 594.0, "completions/min_terminated_length": 594.0, "epoch": 0.0074910083095621974, "grad_norm": 0.0, "learning_rate": 8.495e-07, "loss": 0.0, "num_tokens": 1621561.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 302 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7198.0, "completions/mean_length": 7695.0, "completions/mean_terminated_length": 7198.0, "completions/min_length": 7198.0, "completions/min_terminated_length": 7198.0, "epoch": 0.007515812972838894, "grad_norm": 2.9910032749176025, "learning_rate": 8.489999999999999e-07, "loss": -0.707, "num_tokens": 1629677.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4578.0, "completions/max_terminated_length": 4578.0, "completions/mean_length": 3538.0, "completions/mean_terminated_length": 3538.0, "completions/min_length": 2498.0, "completions/min_terminated_length": 2498.0, "epoch": 0.007540617636115589, "grad_norm": 0.0, "learning_rate": 8.485e-07, "loss": 0.0, "num_tokens": 1637575.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 304 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007565422299392286, "grad_norm": 0.0, "learning_rate": 8.48e-07, "loss": 0.0, "num_tokens": 1638525.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 305 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007590226962668982, "grad_norm": 0.0, "learning_rate": 8.475e-07, "loss": 0.0, "num_tokens": 1639427.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 306 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007615031625945678, "grad_norm": 0.0, "learning_rate": 8.469999999999999e-07, "loss": 0.0, "num_tokens": 1640273.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 307 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7388.0, "completions/max_terminated_length": 7388.0, "completions/mean_length": 6585.5, "completions/mean_terminated_length": 6585.5, "completions/min_length": 5783.0, "completions/min_terminated_length": 5783.0, "epoch": 0.007639836289222374, "grad_norm": 0.0, "learning_rate": 8.465e-07, "loss": 0.0, "num_tokens": 1654344.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 308 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0076646409524990695, "grad_norm": 0.0, "learning_rate": 8.459999999999999e-07, "loss": 0.0, "num_tokens": 1655426.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3345.0, "completions/max_terminated_length": 3345.0, "completions/mean_length": 2752.5, "completions/mean_terminated_length": 2752.5, "completions/min_length": 2160.0, "completions/min_terminated_length": 2160.0, "epoch": 0.007689445615775766, "grad_norm": 0.0, "learning_rate": 8.455e-07, "loss": 0.0, "num_tokens": 1661827.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7136.0, "completions/max_terminated_length": 7136.0, "completions/mean_length": 5850.0, "completions/mean_terminated_length": 5850.0, "completions/min_length": 4564.0, "completions/min_terminated_length": 4564.0, "epoch": 0.0077142502790524614, "grad_norm": 0.0, "learning_rate": 8.45e-07, "loss": 0.0, "num_tokens": 1674373.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2385.0, "completions/max_terminated_length": 2385.0, "completions/mean_length": 2318.5, "completions/mean_terminated_length": 2318.5, "completions/min_length": 2252.0, "completions/min_terminated_length": 2252.0, "epoch": 0.007739054942329158, "grad_norm": 0.0, "learning_rate": 8.445e-07, "loss": 0.0, "num_tokens": 1679810.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7795.0, "completions/mean_length": 7993.5, "completions/mean_terminated_length": 7795.0, "completions/min_length": 7795.0, "completions/min_terminated_length": 7795.0, "epoch": 0.007763859605605854, "grad_norm": 3.0675933361053467, "learning_rate": 8.439999999999999e-07, "loss": -0.707, "num_tokens": 1688497.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 313 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 893.0, "completions/max_terminated_length": 893.0, "completions/mean_length": 679.5, "completions/mean_terminated_length": 679.5, "completions/min_length": 466.0, "completions/min_terminated_length": 466.0, "epoch": 0.00778866426888255, "grad_norm": 0.0, "learning_rate": 8.435e-07, "loss": 0.0, "num_tokens": 1690684.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 314 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007813468932159246, "grad_norm": 0.0, "learning_rate": 8.429999999999999e-07, "loss": 0.0, "num_tokens": 1691694.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 315 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007838273595435942, "grad_norm": 0.0, "learning_rate": 8.425e-07, "loss": 0.0, "num_tokens": 1692582.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7567.0, "completions/mean_length": 7879.5, "completions/mean_terminated_length": 7567.0, "completions/min_length": 7567.0, "completions/min_terminated_length": 7567.0, "epoch": 0.007863078258712637, "grad_norm": 2.8076772689819336, "learning_rate": 8.419999999999999e-07, "loss": -0.707, "num_tokens": 1701095.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 317 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007887882921989334, "grad_norm": 0.0, "learning_rate": 8.415e-07, "loss": 0.0, "num_tokens": 1701979.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 318 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1526.0, "completions/mean_length": 4859.0, "completions/mean_terminated_length": 1526.0, "completions/min_length": 1526.0, "completions/min_terminated_length": 1526.0, "epoch": 0.00791268758526603, "grad_norm": 0.0, "learning_rate": 8.41e-07, "loss": 0.0, "num_tokens": 1704315.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 319 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7678.0, "completions/mean_length": 7935.0, "completions/mean_terminated_length": 7678.0, "completions/min_length": 7678.0, "completions/min_terminated_length": 7678.0, "epoch": 0.007937492248542725, "grad_norm": 3.083261728286743, "learning_rate": 8.404999999999999e-07, "loss": -0.707, "num_tokens": 1712951.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 320 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007962296911819423, "grad_norm": 0.0, "learning_rate": 8.399999999999999e-07, "loss": 0.0, "num_tokens": 1713909.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 321 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.007987101575096118, "grad_norm": 0.0, "learning_rate": 8.395e-07, "loss": 0.0, "num_tokens": 1714777.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 322 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5231.0, "completions/mean_length": 6711.5, "completions/mean_terminated_length": 5231.0, "completions/min_length": 5231.0, "completions/min_terminated_length": 5231.0, "epoch": 0.008011906238372814, "grad_norm": 0.0, "learning_rate": 8.389999999999999e-07, "loss": 0.0, "num_tokens": 1721000.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 323 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.00803671090164951, "grad_norm": 0.0, "learning_rate": 8.385e-07, "loss": 0.0, "num_tokens": 1721954.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 571.0, "completions/max_terminated_length": 571.0, "completions/mean_length": 563.5, "completions/mean_terminated_length": 563.5, "completions/min_length": 556.0, "completions/min_terminated_length": 556.0, "epoch": 0.008061515564926207, "grad_norm": 0.0, "learning_rate": 8.38e-07, "loss": 0.0, "num_tokens": 1723919.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 325 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008086320228202902, "grad_norm": 0.0, "learning_rate": 8.375e-07, "loss": 0.0, "num_tokens": 1724909.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2863.0, "completions/mean_length": 5527.5, "completions/mean_terminated_length": 2863.0, "completions/min_length": 2863.0, "completions/min_terminated_length": 2863.0, "epoch": 0.008111124891479598, "grad_norm": 5.0493245124816895, "learning_rate": 8.369999999999999e-07, "loss": -0.707, "num_tokens": 1728654.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2760.0, "completions/max_terminated_length": 2760.0, "completions/mean_length": 1847.0, "completions/mean_terminated_length": 1847.0, "completions/min_length": 934.0, "completions/min_terminated_length": 934.0, "epoch": 0.008135929554756295, "grad_norm": 0.0, "learning_rate": 8.365e-07, "loss": 0.0, "num_tokens": 1733296.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 328 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3083.0, "completions/max_terminated_length": 3083.0, "completions/mean_length": 2489.5, "completions/mean_terminated_length": 2489.5, "completions/min_length": 1896.0, "completions/min_terminated_length": 1896.0, "epoch": 0.00816073421803299, "grad_norm": 0.0, "learning_rate": 8.359999999999999e-07, "loss": 0.0, "num_tokens": 1739279.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 329 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008185538881309686, "grad_norm": 0.0, "learning_rate": 8.355e-07, "loss": 0.0, "num_tokens": 1740265.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 330 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008210343544586383, "grad_norm": 0.0, "learning_rate": 8.349999999999999e-07, "loss": 0.0, "num_tokens": 1741149.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7080.0, "completions/mean_length": 7636.0, "completions/mean_terminated_length": 7080.0, "completions/min_length": 7080.0, "completions/min_terminated_length": 7080.0, "epoch": 0.008235148207863079, "grad_norm": 0.0, "learning_rate": 8.345e-07, "loss": 0.0, "num_tokens": 1749017.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3986.0, "completions/max_terminated_length": 3986.0, "completions/mean_length": 2922.5, "completions/mean_terminated_length": 2922.5, "completions/min_length": 1859.0, "completions/min_terminated_length": 1859.0, "epoch": 0.008259952871139774, "grad_norm": 0.0, "learning_rate": 8.34e-07, "loss": 0.0, "num_tokens": 1755684.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 333 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2684.0, "completions/max_terminated_length": 2684.0, "completions/mean_length": 2492.0, "completions/mean_terminated_length": 2492.0, "completions/min_length": 2300.0, "completions/min_terminated_length": 2300.0, "epoch": 0.00828475753441647, "grad_norm": 0.0, "learning_rate": 8.334999999999999e-07, "loss": 0.0, "num_tokens": 1761552.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 334 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7504.0, "completions/mean_length": 7848.0, "completions/mean_terminated_length": 7504.0, "completions/min_length": 7504.0, "completions/min_terminated_length": 7504.0, "epoch": 0.008309562197693167, "grad_norm": 0.0, "learning_rate": 8.329999999999999e-07, "loss": 0.0, "num_tokens": 1769990.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 335 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008334366860969862, "grad_norm": 0.0, "learning_rate": 8.325e-07, "loss": 0.0, "num_tokens": 1770930.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5445.0, "completions/mean_length": 6818.5, "completions/mean_terminated_length": 5445.0, "completions/min_length": 5445.0, "completions/min_terminated_length": 5445.0, "epoch": 0.008359171524246558, "grad_norm": 3.492112398147583, "learning_rate": 8.319999999999999e-07, "loss": -0.707, "num_tokens": 1777247.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6479.0, "completions/max_terminated_length": 6479.0, "completions/mean_length": 4587.0, "completions/mean_terminated_length": 4587.0, "completions/min_length": 2695.0, "completions/min_terminated_length": 2695.0, "epoch": 0.008383976187523255, "grad_norm": 2.615051507949829, "learning_rate": 8.315e-07, "loss": 0.2916, "num_tokens": 1787359.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 338 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6765.0, "completions/mean_length": 7478.5, "completions/mean_terminated_length": 6765.0, "completions/min_length": 6765.0, "completions/min_terminated_length": 6765.0, "epoch": 0.00840878085079995, "grad_norm": 0.0, "learning_rate": 8.31e-07, "loss": 0.0, "num_tokens": 1795074.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 339 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008433585514076646, "grad_norm": 0.0, "learning_rate": 8.304999999999999e-07, "loss": 0.0, "num_tokens": 1795914.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2349.0, "completions/max_terminated_length": 2349.0, "completions/mean_length": 1695.0, "completions/mean_terminated_length": 1695.0, "completions/min_length": 1041.0, "completions/min_terminated_length": 1041.0, "epoch": 0.008458390177353342, "grad_norm": 0.0, "learning_rate": 8.299999999999999e-07, "loss": 0.0, "num_tokens": 1800136.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6199.0, "completions/max_terminated_length": 6199.0, "completions/mean_length": 5249.5, "completions/mean_terminated_length": 5249.5, "completions/min_length": 4300.0, "completions/min_terminated_length": 4300.0, "epoch": 0.008483194840630039, "grad_norm": 0.0, "learning_rate": 8.295e-07, "loss": 0.0, "num_tokens": 1811701.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5396.0, "completions/max_terminated_length": 5396.0, "completions/mean_length": 4177.5, "completions/mean_terminated_length": 4177.5, "completions/min_length": 2959.0, "completions/min_terminated_length": 2959.0, "epoch": 0.008507999503906735, "grad_norm": 0.0, "learning_rate": 8.289999999999999e-07, "loss": 0.0, "num_tokens": 1820936.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7670.0, "completions/mean_length": 7931.0, "completions/mean_terminated_length": 7670.0, "completions/min_length": 7670.0, "completions/min_terminated_length": 7670.0, "epoch": 0.00853280416718343, "grad_norm": 0.0, "learning_rate": 8.285e-07, "loss": 0.0, "num_tokens": 1829478.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 344 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008557608830460127, "grad_norm": 0.0, "learning_rate": 8.28e-07, "loss": 0.0, "num_tokens": 1830378.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7961.0, "completions/mean_length": 8076.5, "completions/mean_terminated_length": 7961.0, "completions/min_length": 7961.0, "completions/min_terminated_length": 7961.0, "epoch": 0.008582413493736823, "grad_norm": 3.680582284927368, "learning_rate": 8.275e-07, "loss": -0.707, "num_tokens": 1839185.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2601.0, "completions/max_terminated_length": 2601.0, "completions/mean_length": 2152.0, "completions/mean_terminated_length": 2152.0, "completions/min_length": 1703.0, "completions/min_terminated_length": 1703.0, "epoch": 0.008607218157013518, "grad_norm": 0.0, "learning_rate": 8.269999999999999e-07, "loss": 0.0, "num_tokens": 1844335.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 347 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008632022820290214, "grad_norm": 0.0, "learning_rate": 8.264999999999999e-07, "loss": 0.0, "num_tokens": 1845303.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 348 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008656827483566911, "grad_norm": 0.0, "learning_rate": 8.259999999999999e-07, "loss": 0.0, "num_tokens": 1846267.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5365.0, "completions/mean_length": 6778.5, "completions/mean_terminated_length": 5365.0, "completions/min_length": 5365.0, "completions/min_terminated_length": 5365.0, "epoch": 0.008681632146843607, "grad_norm": 3.445250988006592, "learning_rate": 8.255e-07, "loss": -0.707, "num_tokens": 1852504.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1344.0, "completions/max_terminated_length": 1344.0, "completions/mean_length": 1259.0, "completions/mean_terminated_length": 1259.0, "completions/min_length": 1174.0, "completions/min_terminated_length": 1174.0, "epoch": 0.008706436810120302, "grad_norm": 0.0, "learning_rate": 8.249999999999999e-07, "loss": 0.0, "num_tokens": 1855824.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 351 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 972.0, "completions/max_terminated_length": 972.0, "completions/mean_length": 823.0, "completions/mean_terminated_length": 823.0, "completions/min_length": 674.0, "completions/min_terminated_length": 674.0, "epoch": 0.008731241473397, "grad_norm": 0.0, "learning_rate": 8.245e-07, "loss": 0.0, "num_tokens": 1858254.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 352 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5914.0, "completions/max_terminated_length": 5914.0, "completions/mean_length": 5663.5, "completions/mean_terminated_length": 5663.5, "completions/min_length": 5413.0, "completions/min_terminated_length": 5413.0, "epoch": 0.008756046136673695, "grad_norm": 0.0, "learning_rate": 8.24e-07, "loss": 0.0, "num_tokens": 1870471.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6837.0, "completions/max_terminated_length": 6837.0, "completions/mean_length": 4429.0, "completions/mean_terminated_length": 4429.0, "completions/min_length": 2021.0, "completions/min_terminated_length": 2021.0, "epoch": 0.00878085079995039, "grad_norm": 0.0, "learning_rate": 8.234999999999999e-07, "loss": 0.0, "num_tokens": 1880287.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1436.0, "completions/max_terminated_length": 1436.0, "completions/mean_length": 1142.0, "completions/mean_terminated_length": 1142.0, "completions/min_length": 848.0, "completions/min_terminated_length": 848.0, "epoch": 0.008805655463227086, "grad_norm": 0.0, "learning_rate": 8.229999999999999e-07, "loss": 0.0, "num_tokens": 1883407.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7576.0, "completions/max_terminated_length": 7576.0, "completions/mean_length": 6329.0, "completions/mean_terminated_length": 6329.0, "completions/min_length": 5082.0, "completions/min_terminated_length": 5082.0, "epoch": 0.008830460126503783, "grad_norm": 2.2477328777313232, "learning_rate": 8.225e-07, "loss": 0.1393, "num_tokens": 1897019.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 356 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1638.0, "completions/max_terminated_length": 1638.0, "completions/mean_length": 1588.5, "completions/mean_terminated_length": 1588.5, "completions/min_length": 1539.0, "completions/min_terminated_length": 1539.0, "epoch": 0.008855264789780479, "grad_norm": 0.0, "learning_rate": 8.219999999999999e-07, "loss": 0.0, "num_tokens": 1901128.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 357 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008880069453057174, "grad_norm": 0.0, "learning_rate": 8.215e-07, "loss": 0.0, "num_tokens": 1902110.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 358 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008904874116333871, "grad_norm": 0.0, "learning_rate": 8.21e-07, "loss": 0.0, "num_tokens": 1902948.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 359 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6832.0, "completions/max_terminated_length": 6832.0, "completions/mean_length": 4367.5, "completions/mean_terminated_length": 4367.5, "completions/min_length": 1903.0, "completions/min_terminated_length": 1903.0, "epoch": 0.008929678779610567, "grad_norm": 0.0, "learning_rate": 8.205e-07, "loss": 0.0, "num_tokens": 1912671.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2399.0, "completions/max_terminated_length": 2399.0, "completions/mean_length": 1849.5, "completions/mean_terminated_length": 1849.5, "completions/min_length": 1300.0, "completions/min_terminated_length": 1300.0, "epoch": 0.008954483442887263, "grad_norm": 0.0, "learning_rate": 8.199999999999999e-07, "loss": 0.0, "num_tokens": 1917374.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 361 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.008979288106163958, "grad_norm": 0.0, "learning_rate": 8.194999999999999e-07, "loss": 0.0, "num_tokens": 1918326.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4429.0, "completions/mean_length": 6310.5, "completions/mean_terminated_length": 4429.0, "completions/min_length": 4429.0, "completions/min_terminated_length": 4429.0, "epoch": 0.009004092769440655, "grad_norm": 0.0, "learning_rate": 8.189999999999999e-07, "loss": 0.0, "num_tokens": 1923741.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5487.0, "completions/max_terminated_length": 5487.0, "completions/mean_length": 3559.5, "completions/mean_terminated_length": 3559.5, "completions/min_length": 1632.0, "completions/min_terminated_length": 1632.0, "epoch": 0.00902889743271735, "grad_norm": 0.0, "learning_rate": 8.185e-07, "loss": 0.0, "num_tokens": 1931748.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1211.0, "completions/max_terminated_length": 1211.0, "completions/mean_length": 1026.0, "completions/mean_terminated_length": 1026.0, "completions/min_length": 841.0, "completions/min_terminated_length": 841.0, "epoch": 0.009053702095994046, "grad_norm": 0.0, "learning_rate": 8.179999999999999e-07, "loss": 0.0, "num_tokens": 1934828.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3969.0, "completions/mean_length": 6080.5, "completions/mean_terminated_length": 3969.0, "completions/min_length": 3969.0, "completions/min_terminated_length": 3969.0, "epoch": 0.009078506759270744, "grad_norm": 4.084251880645752, "learning_rate": 8.175e-07, "loss": -0.707, "num_tokens": 1939887.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1763.0, "completions/max_terminated_length": 1763.0, "completions/mean_length": 1425.0, "completions/mean_terminated_length": 1425.0, "completions/min_length": 1087.0, "completions/min_terminated_length": 1087.0, "epoch": 0.009103311422547439, "grad_norm": 0.0, "learning_rate": 8.169999999999999e-07, "loss": 0.0, "num_tokens": 1943541.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 367 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1165.0, "completions/max_terminated_length": 1165.0, "completions/mean_length": 1052.5, "completions/mean_terminated_length": 1052.5, "completions/min_length": 940.0, "completions/min_terminated_length": 940.0, "epoch": 0.009128116085824135, "grad_norm": 5.068020343780518, "learning_rate": 8.164999999999999e-07, "loss": 0.0756, "num_tokens": 1946454.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 368 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.00915292074910083, "grad_norm": 0.0, "learning_rate": 8.159999999999999e-07, "loss": 0.0, "num_tokens": 1947432.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7847.0, "completions/max_terminated_length": 7847.0, "completions/mean_length": 6855.0, "completions/mean_terminated_length": 6855.0, "completions/min_length": 5863.0, "completions/min_terminated_length": 5863.0, "epoch": 0.009177725412377527, "grad_norm": 0.0, "learning_rate": 8.155e-07, "loss": 0.0, "num_tokens": 1962054.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 370 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2755.0, "completions/max_terminated_length": 2755.0, "completions/mean_length": 2182.0, "completions/mean_terminated_length": 2182.0, "completions/min_length": 1609.0, "completions/min_terminated_length": 1609.0, "epoch": 0.009202530075654223, "grad_norm": 0.0, "learning_rate": 8.149999999999999e-07, "loss": 0.0, "num_tokens": 1967256.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1421.0, "completions/max_terminated_length": 1421.0, "completions/mean_length": 1072.0, "completions/mean_terminated_length": 1072.0, "completions/min_length": 723.0, "completions/min_terminated_length": 723.0, "epoch": 0.009227334738930918, "grad_norm": 6.202691078186035, "learning_rate": 8.145e-07, "loss": 0.2302, "num_tokens": 1970212.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 372 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009252139402207616, "grad_norm": 0.0, "learning_rate": 8.14e-07, "loss": 0.0, "num_tokens": 1971064.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 373 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6027.0, "completions/mean_length": 7109.5, "completions/mean_terminated_length": 6027.0, "completions/min_length": 6027.0, "completions/min_terminated_length": 6027.0, "epoch": 0.009276944065484311, "grad_norm": 3.555191993713379, "learning_rate": 8.134999999999999e-07, "loss": -0.707, "num_tokens": 1978041.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 374 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009301748728761007, "grad_norm": 0.0, "learning_rate": 8.129999999999999e-07, "loss": 0.0, "num_tokens": 1978955.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 375 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009326553392037702, "grad_norm": 0.0, "learning_rate": 8.125e-07, "loss": 0.0, "num_tokens": 1979817.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 376 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0093513580553144, "grad_norm": 0.0, "learning_rate": 8.12e-07, "loss": 0.0, "num_tokens": 1980821.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 377 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4373.0, "completions/max_terminated_length": 4373.0, "completions/mean_length": 4059.5, "completions/mean_terminated_length": 4059.5, "completions/min_length": 3746.0, "completions/min_terminated_length": 3746.0, "epoch": 0.009376162718591095, "grad_norm": 0.0, "learning_rate": 8.115e-07, "loss": 0.0, "num_tokens": 1989810.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 378 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7947.0, "completions/mean_length": 8069.5, "completions/mean_terminated_length": 7947.0, "completions/min_length": 7947.0, "completions/min_terminated_length": 7947.0, "epoch": 0.00940096738186779, "grad_norm": 0.0, "learning_rate": 8.11e-07, "loss": 0.0, "num_tokens": 1998725.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7018.0, "completions/mean_length": 7605.0, "completions/mean_terminated_length": 7018.0, "completions/min_length": 7018.0, "completions/min_terminated_length": 7018.0, "epoch": 0.009425772045144488, "grad_norm": 0.0, "learning_rate": 8.105e-07, "loss": 0.0, "num_tokens": 2006617.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3111.0, "completions/max_terminated_length": 3111.0, "completions/mean_length": 2316.5, "completions/mean_terminated_length": 2316.5, "completions/min_length": 1522.0, "completions/min_terminated_length": 1522.0, "epoch": 0.009450576708421183, "grad_norm": 0.0, "learning_rate": 8.1e-07, "loss": 0.0, "num_tokens": 2012106.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 381 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009475381371697879, "grad_norm": 0.0, "learning_rate": 8.094999999999999e-07, "loss": 0.0, "num_tokens": 2013178.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4833.0, "completions/max_terminated_length": 4833.0, "completions/mean_length": 3526.5, "completions/mean_terminated_length": 3526.5, "completions/min_length": 2220.0, "completions/min_terminated_length": 2220.0, "epoch": 0.009500186034974576, "grad_norm": 0.0, "learning_rate": 8.09e-07, "loss": 0.0, "num_tokens": 2021143.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 383 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009524990698251272, "grad_norm": 0.0, "learning_rate": 8.085e-07, "loss": 0.0, "num_tokens": 2021985.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 384 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009549795361527967, "grad_norm": 0.0, "learning_rate": 8.08e-07, "loss": 0.0, "num_tokens": 2023159.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7713.0, "completions/max_terminated_length": 7713.0, "completions/mean_length": 7167.0, "completions/mean_terminated_length": 7167.0, "completions/min_length": 6621.0, "completions/min_terminated_length": 6621.0, "epoch": 0.009574600024804663, "grad_norm": 2.5286412239074707, "learning_rate": 8.075e-07, "loss": 0.0539, "num_tokens": 2038433.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6625.0, "completions/max_terminated_length": 6625.0, "completions/mean_length": 4393.5, "completions/mean_terminated_length": 4393.5, "completions/min_length": 2162.0, "completions/min_terminated_length": 2162.0, "epoch": 0.00959940468808136, "grad_norm": 0.0, "learning_rate": 8.070000000000001e-07, "loss": 0.0, "num_tokens": 2048062.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 387 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3394.0, "completions/mean_length": 5793.0, "completions/mean_terminated_length": 3394.0, "completions/min_length": 3394.0, "completions/min_terminated_length": 3394.0, "epoch": 0.009624209351358055, "grad_norm": 0.0, "learning_rate": 8.064999999999999e-07, "loss": 0.0, "num_tokens": 2052302.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 388 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009649014014634751, "grad_norm": 0.0, "learning_rate": 8.06e-07, "loss": 0.0, "num_tokens": 2053292.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5483.0, "completions/mean_length": 6837.5, "completions/mean_terminated_length": 5483.0, "completions/min_length": 5483.0, "completions/min_terminated_length": 5483.0, "epoch": 0.009673818677911448, "grad_norm": 3.4894680976867676, "learning_rate": 8.055e-07, "loss": -0.707, "num_tokens": 2059633.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 390 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7898.0, "completions/max_terminated_length": 7898.0, "completions/mean_length": 7623.0, "completions/mean_terminated_length": 7623.0, "completions/min_length": 7348.0, "completions/min_terminated_length": 7348.0, "epoch": 0.009698623341188144, "grad_norm": 2.3468575477600098, "learning_rate": 8.05e-07, "loss": 0.0255, "num_tokens": 2076427.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 391 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1778.0, "completions/max_terminated_length": 1778.0, "completions/mean_length": 1444.5, "completions/mean_terminated_length": 1444.5, "completions/min_length": 1111.0, "completions/min_terminated_length": 1111.0, "epoch": 0.00972342800446484, "grad_norm": 0.0, "learning_rate": 8.045e-07, "loss": 0.0, "num_tokens": 2080182.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7101.0, "completions/mean_length": 7646.5, "completions/mean_terminated_length": 7101.0, "completions/min_length": 7101.0, "completions/min_terminated_length": 7101.0, "epoch": 0.009748232667741535, "grad_norm": 2.8189241886138916, "learning_rate": 8.04e-07, "loss": -0.707, "num_tokens": 2088173.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 393 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009773037331018232, "grad_norm": 0.0, "learning_rate": 8.034999999999999e-07, "loss": 0.0, "num_tokens": 2089189.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5650.0, "completions/mean_length": 6921.0, "completions/mean_terminated_length": 5650.0, "completions/min_length": 5650.0, "completions/min_terminated_length": 5650.0, "epoch": 0.009797841994294927, "grad_norm": 4.324097633361816, "learning_rate": 8.03e-07, "loss": -0.707, "num_tokens": 2095845.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3865.0, "completions/mean_length": 6028.5, "completions/mean_terminated_length": 3865.0, "completions/min_length": 3865.0, "completions/min_terminated_length": 3865.0, "epoch": 0.009822646657571623, "grad_norm": 4.358633041381836, "learning_rate": 8.024999999999999e-07, "loss": -0.707, "num_tokens": 2100586.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6539.0, "completions/mean_length": 7365.5, "completions/mean_terminated_length": 6539.0, "completions/min_length": 6539.0, "completions/min_terminated_length": 6539.0, "epoch": 0.00984745132084832, "grad_norm": 3.9339897632598877, "learning_rate": 8.02e-07, "loss": -0.707, "num_tokens": 2108011.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 397 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8158.0, "completions/max_terminated_length": 8158.0, "completions/mean_length": 7863.0, "completions/mean_terminated_length": 7863.0, "completions/min_length": 7568.0, "completions/min_terminated_length": 7568.0, "epoch": 0.009872255984125016, "grad_norm": 0.0, "learning_rate": 8.015e-07, "loss": 0.0, "num_tokens": 2124605.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 398 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009897060647401711, "grad_norm": 0.0, "learning_rate": 8.01e-07, "loss": 0.0, "num_tokens": 2125547.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 399 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7757.0, "completions/max_terminated_length": 7757.0, "completions/mean_length": 5184.5, "completions/mean_terminated_length": 5184.5, "completions/min_length": 2612.0, "completions/min_terminated_length": 2612.0, "epoch": 0.009921865310678407, "grad_norm": 2.350358486175537, "learning_rate": 8.005e-07, "loss": 0.3508, "num_tokens": 2136754.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 400 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.009946669973955104, "grad_norm": 0.0, "learning_rate": 8e-07, "loss": 0.0, "num_tokens": 2137636.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6249.0, "completions/max_terminated_length": 6249.0, "completions/mean_length": 4156.5, "completions/mean_terminated_length": 4156.5, "completions/min_length": 2064.0, "completions/min_terminated_length": 2064.0, "epoch": 0.0099714746372318, "grad_norm": 0.0, "learning_rate": 7.994999999999999e-07, "loss": 0.0, "num_tokens": 2146827.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5805.0, "completions/max_terminated_length": 5805.0, "completions/mean_length": 4778.5, "completions/mean_terminated_length": 4778.5, "completions/min_length": 3752.0, "completions/min_terminated_length": 3752.0, "epoch": 0.009996279300508495, "grad_norm": 0.0, "learning_rate": 7.99e-07, "loss": 0.0, "num_tokens": 2157266.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 403 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010021083963785192, "grad_norm": 0.0, "learning_rate": 7.985e-07, "loss": 0.0, "num_tokens": 2158130.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2740.0, "completions/max_terminated_length": 2740.0, "completions/mean_length": 2163.5, "completions/mean_terminated_length": 2163.5, "completions/min_length": 1587.0, "completions/min_terminated_length": 1587.0, "epoch": 0.010045888627061888, "grad_norm": 3.8211476802825928, "learning_rate": 7.98e-07, "loss": 0.1884, "num_tokens": 2163387.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 405 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010070693290338583, "grad_norm": 0.0, "learning_rate": 7.975e-07, "loss": 0.0, "num_tokens": 2164283.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1604.0, "completions/max_terminated_length": 1604.0, "completions/mean_length": 1116.0, "completions/mean_terminated_length": 1116.0, "completions/min_length": 628.0, "completions/min_terminated_length": 628.0, "epoch": 0.010095497953615279, "grad_norm": 0.0, "learning_rate": 7.970000000000001e-07, "loss": 0.0, "num_tokens": 2167391.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 407 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010120302616891976, "grad_norm": 0.0, "learning_rate": 7.964999999999999e-07, "loss": 0.0, "num_tokens": 2168277.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3535.0, "completions/mean_length": 5863.5, "completions/mean_terminated_length": 3535.0, "completions/min_length": 3535.0, "completions/min_terminated_length": 3535.0, "epoch": 0.010145107280168672, "grad_norm": 0.0, "learning_rate": 7.96e-07, "loss": 0.0, "num_tokens": 2172658.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4986.0, "completions/mean_length": 6589.0, "completions/mean_terminated_length": 4986.0, "completions/min_length": 4986.0, "completions/min_terminated_length": 4986.0, "epoch": 0.010169911943445367, "grad_norm": 0.0, "learning_rate": 7.954999999999999e-07, "loss": 0.0, "num_tokens": 2178500.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3955.0, "completions/max_terminated_length": 3955.0, "completions/mean_length": 2778.0, "completions/mean_terminated_length": 2778.0, "completions/min_length": 1601.0, "completions/min_terminated_length": 1601.0, "epoch": 0.010194716606722064, "grad_norm": 0.0, "learning_rate": 7.95e-07, "loss": 0.0, "num_tokens": 2184988.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 411 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7213.0, "completions/max_terminated_length": 7213.0, "completions/mean_length": 6134.5, "completions/mean_terminated_length": 6134.5, "completions/min_length": 5056.0, "completions/min_terminated_length": 5056.0, "epoch": 0.01021952126999876, "grad_norm": 2.6535942554473877, "learning_rate": 7.945e-07, "loss": -0.1243, "num_tokens": 2198443.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5567.0, "completions/max_terminated_length": 5567.0, "completions/mean_length": 4780.0, "completions/mean_terminated_length": 4780.0, "completions/min_length": 3993.0, "completions/min_terminated_length": 3993.0, "epoch": 0.010244325933275455, "grad_norm": 0.0, "learning_rate": 7.94e-07, "loss": 0.0, "num_tokens": 2208855.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 413 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010269130596552151, "grad_norm": 0.0, "learning_rate": 7.934999999999999e-07, "loss": 0.0, "num_tokens": 2209725.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2033.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 2032.0, "completions/mean_terminated_length": 2032.0, "completions/min_length": 2031.0, "completions/min_terminated_length": 2031.0, "epoch": 0.010293935259828848, "grad_norm": 0.0, "learning_rate": 7.93e-07, "loss": 0.0, "num_tokens": 2214599.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 415 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010318739923105544, "grad_norm": 0.0, "learning_rate": 7.924999999999999e-07, "loss": 0.0, "num_tokens": 2215561.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1082.0, "completions/max_terminated_length": 1082.0, "completions/mean_length": 1037.0, "completions/mean_terminated_length": 1037.0, "completions/min_length": 992.0, "completions/min_terminated_length": 992.0, "epoch": 0.01034354458638224, "grad_norm": 0.0, "learning_rate": 7.92e-07, "loss": 0.0, "num_tokens": 2218449.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 417 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7746.0, "completions/max_terminated_length": 7746.0, "completions/mean_length": 6885.5, "completions/mean_terminated_length": 6885.5, "completions/min_length": 6025.0, "completions/min_terminated_length": 6025.0, "epoch": 0.010368349249658937, "grad_norm": 0.0, "learning_rate": 7.915e-07, "loss": 0.0, "num_tokens": 2233178.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5944.0, "completions/max_terminated_length": 5944.0, "completions/mean_length": 4624.5, "completions/mean_terminated_length": 4624.5, "completions/min_length": 3305.0, "completions/min_terminated_length": 3305.0, "epoch": 0.010393153912935632, "grad_norm": 0.0, "learning_rate": 7.91e-07, "loss": 0.0, "num_tokens": 2243327.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 419 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010417958576212328, "grad_norm": 0.0, "learning_rate": 7.905e-07, "loss": 0.0, "num_tokens": 2244373.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7181.0, "completions/max_terminated_length": 7181.0, "completions/mean_length": 5977.5, "completions/mean_terminated_length": 5977.5, "completions/min_length": 4774.0, "completions/min_terminated_length": 4774.0, "epoch": 0.010442763239489023, "grad_norm": 0.0, "learning_rate": 7.9e-07, "loss": 0.0, "num_tokens": 2258376.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 421 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4115.0, "completions/mean_length": 6153.5, "completions/mean_terminated_length": 4115.0, "completions/min_length": 4115.0, "completions/min_terminated_length": 4115.0, "epoch": 0.01046756790276572, "grad_norm": 4.545650482177734, "learning_rate": 7.894999999999999e-07, "loss": -0.707, "num_tokens": 2263389.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7541.0, "completions/max_terminated_length": 7541.0, "completions/mean_length": 6830.5, "completions/mean_terminated_length": 6830.5, "completions/min_length": 6120.0, "completions/min_terminated_length": 6120.0, "epoch": 0.010492372566042416, "grad_norm": 0.0, "learning_rate": 7.89e-07, "loss": 0.0, "num_tokens": 2277876.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1365.0, "completions/max_terminated_length": 1365.0, "completions/mean_length": 1152.5, "completions/mean_terminated_length": 1152.5, "completions/min_length": 940.0, "completions/min_terminated_length": 940.0, "epoch": 0.010517177229319111, "grad_norm": 0.0, "learning_rate": 7.884999999999999e-07, "loss": 0.0, "num_tokens": 2281003.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 424 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010541981892595809, "grad_norm": 0.0, "learning_rate": 7.88e-07, "loss": 0.0, "num_tokens": 2281927.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6938.0, "completions/max_terminated_length": 6938.0, "completions/mean_length": 5279.0, "completions/mean_terminated_length": 5279.0, "completions/min_length": 3620.0, "completions/min_terminated_length": 3620.0, "epoch": 0.010566786555872504, "grad_norm": 2.412282943725586, "learning_rate": 7.875e-07, "loss": -0.2222, "num_tokens": 2293715.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7797.0, "completions/max_terminated_length": 7797.0, "completions/mean_length": 6563.5, "completions/mean_terminated_length": 6563.5, "completions/min_length": 5330.0, "completions/min_terminated_length": 5330.0, "epoch": 0.0105915912191492, "grad_norm": 0.0, "learning_rate": 7.87e-07, "loss": 0.0, "num_tokens": 2307728.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5308.0, "completions/max_terminated_length": 5308.0, "completions/mean_length": 4963.5, "completions/mean_terminated_length": 4963.5, "completions/min_length": 4619.0, "completions/min_terminated_length": 4619.0, "epoch": 0.010616395882425897, "grad_norm": 2.9007623195648193, "learning_rate": 7.864999999999999e-07, "loss": -0.0491, "num_tokens": 2318527.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3395.0, "completions/mean_length": 5793.5, "completions/mean_terminated_length": 3395.0, "completions/min_length": 3395.0, "completions/min_terminated_length": 3395.0, "epoch": 0.010641200545702592, "grad_norm": 4.785800457000732, "learning_rate": 7.86e-07, "loss": -0.707, "num_tokens": 2322742.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5390.0, "completions/max_terminated_length": 5390.0, "completions/mean_length": 5118.5, "completions/mean_terminated_length": 5118.5, "completions/min_length": 4847.0, "completions/min_terminated_length": 4847.0, "epoch": 0.010666005208979288, "grad_norm": 0.0, "learning_rate": 7.854999999999999e-07, "loss": 0.0, "num_tokens": 2333899.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4947.0, "completions/max_terminated_length": 4947.0, "completions/mean_length": 4855.5, "completions/mean_terminated_length": 4855.5, "completions/min_length": 4764.0, "completions/min_terminated_length": 4764.0, "epoch": 0.010690809872255983, "grad_norm": 0.0, "learning_rate": 7.85e-07, "loss": 0.0, "num_tokens": 2344500.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 431 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7410.0, "completions/mean_length": 7801.0, "completions/mean_terminated_length": 7410.0, "completions/min_length": 7410.0, "completions/min_terminated_length": 7410.0, "epoch": 0.01071561453553268, "grad_norm": 0.0, "learning_rate": 7.845e-07, "loss": 0.0, "num_tokens": 2352760.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5941.0, "completions/max_terminated_length": 5941.0, "completions/mean_length": 4307.5, "completions/mean_terminated_length": 4307.5, "completions/min_length": 2674.0, "completions/min_terminated_length": 2674.0, "epoch": 0.010740419198809376, "grad_norm": 0.0, "learning_rate": 7.84e-07, "loss": 0.0, "num_tokens": 2362323.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3320.0, "completions/max_terminated_length": 3320.0, "completions/mean_length": 2522.5, "completions/mean_terminated_length": 2522.5, "completions/min_length": 1725.0, "completions/min_terminated_length": 1725.0, "epoch": 0.010765223862086072, "grad_norm": 0.0, "learning_rate": 7.834999999999999e-07, "loss": 0.0, "num_tokens": 2368216.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 434 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010790028525362769, "grad_norm": 0.0, "learning_rate": 7.83e-07, "loss": 0.0, "num_tokens": 2369088.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7761.0, "completions/max_terminated_length": 7761.0, "completions/mean_length": 7102.5, "completions/mean_terminated_length": 7102.5, "completions/min_length": 6444.0, "completions/min_terminated_length": 6444.0, "epoch": 0.010814833188639465, "grad_norm": 0.0, "learning_rate": 7.824999999999999e-07, "loss": 0.0, "num_tokens": 2384237.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5147.0, "completions/mean_length": 6669.5, "completions/mean_terminated_length": 5147.0, "completions/min_length": 5147.0, "completions/min_terminated_length": 5147.0, "epoch": 0.01083963785191616, "grad_norm": 3.332291603088379, "learning_rate": 7.82e-07, "loss": -0.707, "num_tokens": 2390300.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 437 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010864442515192856, "grad_norm": 0.0, "learning_rate": 7.815e-07, "loss": 0.0, "num_tokens": 2391210.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1621.0, "completions/max_terminated_length": 1621.0, "completions/mean_length": 1382.0, "completions/mean_terminated_length": 1382.0, "completions/min_length": 1143.0, "completions/min_terminated_length": 1143.0, "epoch": 0.010889247178469553, "grad_norm": 0.0, "learning_rate": 7.81e-07, "loss": 0.0, "num_tokens": 2394754.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1176.0, "completions/max_terminated_length": 1176.0, "completions/mean_length": 1114.0, "completions/mean_terminated_length": 1114.0, "completions/min_length": 1052.0, "completions/min_terminated_length": 1052.0, "epoch": 0.010914051841746248, "grad_norm": 0.0, "learning_rate": 7.805e-07, "loss": 0.0, "num_tokens": 2397792.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2308.0, "completions/mean_length": 5250.0, "completions/mean_terminated_length": 2308.0, "completions/min_length": 2308.0, "completions/min_terminated_length": 2308.0, "epoch": 0.010938856505022944, "grad_norm": 0.0, "learning_rate": 7.799999999999999e-07, "loss": 0.0, "num_tokens": 2401158.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 441 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010963661168299641, "grad_norm": 0.0, "learning_rate": 7.794999999999999e-07, "loss": 0.0, "num_tokens": 2402032.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 442 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.010988465831576337, "grad_norm": 0.0, "learning_rate": 7.79e-07, "loss": 0.0, "num_tokens": 2402966.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6856.0, "completions/max_terminated_length": 6856.0, "completions/mean_length": 4848.5, "completions/mean_terminated_length": 4848.5, "completions/min_length": 2841.0, "completions/min_terminated_length": 2841.0, "epoch": 0.011013270494853032, "grad_norm": 2.4393889904022217, "learning_rate": 7.784999999999999e-07, "loss": 0.2927, "num_tokens": 2413513.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 444 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.011038075158129728, "grad_norm": 0.0, "learning_rate": 7.78e-07, "loss": 0.0, "num_tokens": 2414507.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1100.0, "completions/max_terminated_length": 1100.0, "completions/mean_length": 913.5, "completions/mean_terminated_length": 913.5, "completions/min_length": 727.0, "completions/min_terminated_length": 727.0, "epoch": 0.011062879821406425, "grad_norm": 0.0, "learning_rate": 7.775e-07, "loss": 0.0, "num_tokens": 2417222.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4343.0, "completions/max_terminated_length": 4343.0, "completions/mean_length": 3563.5, "completions/mean_terminated_length": 3563.5, "completions/min_length": 2784.0, "completions/min_terminated_length": 2784.0, "epoch": 0.01108768448468312, "grad_norm": 0.0, "learning_rate": 7.77e-07, "loss": 0.0, "num_tokens": 2425229.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8065.0, "completions/mean_length": 8128.5, "completions/mean_terminated_length": 8065.0, "completions/min_length": 8065.0, "completions/min_terminated_length": 8065.0, "epoch": 0.011112489147959816, "grad_norm": 2.628190517425537, "learning_rate": 7.764999999999999e-07, "loss": -0.707, "num_tokens": 2434142.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3086.0, "completions/max_terminated_length": 3086.0, "completions/mean_length": 2449.0, "completions/mean_terminated_length": 2449.0, "completions/min_length": 1812.0, "completions/min_terminated_length": 1812.0, "epoch": 0.011137293811236513, "grad_norm": 0.0, "learning_rate": 7.76e-07, "loss": 0.0, "num_tokens": 2440232.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5698.0, "completions/max_terminated_length": 5698.0, "completions/mean_length": 4237.0, "completions/mean_terminated_length": 4237.0, "completions/min_length": 2776.0, "completions/min_terminated_length": 2776.0, "epoch": 0.011162098474513209, "grad_norm": 0.0, "learning_rate": 7.754999999999999e-07, "loss": 0.0, "num_tokens": 2449556.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5616.0, "completions/mean_length": 6904.0, "completions/mean_terminated_length": 5616.0, "completions/min_length": 5616.0, "completions/min_terminated_length": 5616.0, "epoch": 0.011186903137789904, "grad_norm": 3.102053642272949, "learning_rate": 7.75e-07, "loss": -0.707, "num_tokens": 2456046.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1975.0, "completions/max_terminated_length": 1975.0, "completions/mean_length": 1373.5, "completions/mean_terminated_length": 1373.5, "completions/min_length": 772.0, "completions/min_terminated_length": 772.0, "epoch": 0.0112117078010666, "grad_norm": 0.0, "learning_rate": 7.745e-07, "loss": 0.0, "num_tokens": 2459617.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6737.0, "completions/max_terminated_length": 6737.0, "completions/mean_length": 5632.0, "completions/mean_terminated_length": 5632.0, "completions/min_length": 4527.0, "completions/min_terminated_length": 4527.0, "epoch": 0.011236512464343297, "grad_norm": 2.140373945236206, "learning_rate": 7.74e-07, "loss": 0.1387, "num_tokens": 2471995.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 453 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4732.0, "completions/max_terminated_length": 4732.0, "completions/mean_length": 3908.5, "completions/mean_terminated_length": 3908.5, "completions/min_length": 3085.0, "completions/min_terminated_length": 3085.0, "epoch": 0.011261317127619993, "grad_norm": 0.0, "learning_rate": 7.734999999999999e-07, "loss": 0.0, "num_tokens": 2480674.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4519.0, "completions/max_terminated_length": 4519.0, "completions/mean_length": 3874.5, "completions/mean_terminated_length": 3874.5, "completions/min_length": 3230.0, "completions/min_terminated_length": 3230.0, "epoch": 0.011286121790896688, "grad_norm": 4.595510959625244, "learning_rate": 7.729999999999999e-07, "loss": 0.1176, "num_tokens": 2489371.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 455 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.011310926454173385, "grad_norm": 0.0, "learning_rate": 7.724999999999999e-07, "loss": 0.0, "num_tokens": 2490245.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5253.0, "completions/max_terminated_length": 5253.0, "completions/mean_length": 4490.0, "completions/mean_terminated_length": 4490.0, "completions/min_length": 3727.0, "completions/min_terminated_length": 3727.0, "epoch": 0.01133573111745008, "grad_norm": 0.0, "learning_rate": 7.72e-07, "loss": 0.0, "num_tokens": 2500089.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6044.0, "completions/max_terminated_length": 6044.0, "completions/mean_length": 5055.5, "completions/mean_terminated_length": 5055.5, "completions/min_length": 4067.0, "completions/min_terminated_length": 4067.0, "epoch": 0.011360535780726776, "grad_norm": 0.0, "learning_rate": 7.714999999999999e-07, "loss": 0.0, "num_tokens": 2511026.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 458 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1196.0, "completions/max_terminated_length": 1196.0, "completions/mean_length": 1092.5, "completions/mean_terminated_length": 1092.5, "completions/min_length": 989.0, "completions/min_terminated_length": 989.0, "epoch": 0.011385340444003472, "grad_norm": 0.0, "learning_rate": 7.71e-07, "loss": 0.0, "num_tokens": 2514027.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1430.0, "completions/max_terminated_length": 1430.0, "completions/mean_length": 1255.0, "completions/mean_terminated_length": 1255.0, "completions/min_length": 1080.0, "completions/min_terminated_length": 1080.0, "epoch": 0.01141014510728017, "grad_norm": 0.0, "learning_rate": 7.705e-07, "loss": 0.0, "num_tokens": 2517337.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6608.0, "completions/mean_length": 7400.0, "completions/mean_terminated_length": 6608.0, "completions/min_length": 6608.0, "completions/min_terminated_length": 6608.0, "epoch": 0.011434949770556865, "grad_norm": 3.5592448711395264, "learning_rate": 7.699999999999999e-07, "loss": -0.707, "num_tokens": 2524827.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 461 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01145975443383356, "grad_norm": 0.0, "learning_rate": 7.694999999999999e-07, "loss": 0.0, "num_tokens": 2525779.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 462 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.011484559097110257, "grad_norm": 0.0, "learning_rate": 7.69e-07, "loss": 0.0, "num_tokens": 2526749.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 463 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1705.0, "completions/max_terminated_length": 1705.0, "completions/mean_length": 1491.5, "completions/mean_terminated_length": 1491.5, "completions/min_length": 1278.0, "completions/min_terminated_length": 1278.0, "epoch": 0.011509363760386953, "grad_norm": 0.0, "learning_rate": 7.684999999999999e-07, "loss": 0.0, "num_tokens": 2530592.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5379.0, "completions/max_terminated_length": 5379.0, "completions/mean_length": 4619.0, "completions/mean_terminated_length": 4619.0, "completions/min_length": 3859.0, "completions/min_terminated_length": 3859.0, "epoch": 0.011534168423663648, "grad_norm": 0.0, "learning_rate": 7.68e-07, "loss": 0.0, "num_tokens": 2540762.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2367.0, "completions/max_terminated_length": 2367.0, "completions/mean_length": 2360.0, "completions/mean_terminated_length": 2360.0, "completions/min_length": 2353.0, "completions/min_terminated_length": 2353.0, "epoch": 0.011558973086940344, "grad_norm": 0.0, "learning_rate": 7.675e-07, "loss": 0.0, "num_tokens": 2546328.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7318.0, "completions/mean_length": 7755.0, "completions/mean_terminated_length": 7318.0, "completions/min_length": 7318.0, "completions/min_terminated_length": 7318.0, "epoch": 0.011583777750217041, "grad_norm": 2.8636510372161865, "learning_rate": 7.67e-07, "loss": -0.707, "num_tokens": 2554686.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2682.0, "completions/max_terminated_length": 2682.0, "completions/mean_length": 2299.5, "completions/mean_terminated_length": 2299.5, "completions/min_length": 1917.0, "completions/min_terminated_length": 1917.0, "epoch": 0.011608582413493737, "grad_norm": 0.0, "learning_rate": 7.664999999999999e-07, "loss": 0.0, "num_tokens": 2560249.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6236.0, "completions/max_terminated_length": 6236.0, "completions/mean_length": 4895.0, "completions/mean_terminated_length": 4895.0, "completions/min_length": 3554.0, "completions/min_terminated_length": 3554.0, "epoch": 0.011633387076770432, "grad_norm": 0.0, "learning_rate": 7.66e-07, "loss": 0.0, "num_tokens": 2570879.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4921.0, "completions/max_terminated_length": 4921.0, "completions/mean_length": 4782.5, "completions/mean_terminated_length": 4782.5, "completions/min_length": 4644.0, "completions/min_terminated_length": 4644.0, "epoch": 0.01165819174004713, "grad_norm": 0.0, "learning_rate": 7.654999999999999e-07, "loss": 0.0, "num_tokens": 2581442.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 470 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.011682996403323825, "grad_norm": 0.0, "learning_rate": 7.65e-07, "loss": 0.0, "num_tokens": 2582408.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 471 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4896.0, "completions/mean_length": 6544.0, "completions/mean_terminated_length": 4896.0, "completions/min_length": 4896.0, "completions/min_terminated_length": 4896.0, "epoch": 0.01170780106660052, "grad_norm": 3.3537471294403076, "learning_rate": 7.644999999999999e-07, "loss": -0.707, "num_tokens": 2588250.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7651.0, "completions/max_terminated_length": 7651.0, "completions/mean_length": 7252.5, "completions/mean_terminated_length": 7252.5, "completions/min_length": 6854.0, "completions/min_terminated_length": 6854.0, "epoch": 0.011732605729877216, "grad_norm": 0.0, "learning_rate": 7.64e-07, "loss": 0.0, "num_tokens": 2603843.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7038.0, "completions/max_terminated_length": 7038.0, "completions/mean_length": 5863.5, "completions/mean_terminated_length": 5863.5, "completions/min_length": 4689.0, "completions/min_terminated_length": 4689.0, "epoch": 0.011757410393153913, "grad_norm": 0.0, "learning_rate": 7.635e-07, "loss": 0.0, "num_tokens": 2616396.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 474 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.011782215056430609, "grad_norm": 0.0, "learning_rate": 7.629999999999999e-07, "loss": 0.0, "num_tokens": 2617320.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4787.0, "completions/max_terminated_length": 4787.0, "completions/mean_length": 4674.0, "completions/mean_terminated_length": 4674.0, "completions/min_length": 4561.0, "completions/min_terminated_length": 4561.0, "epoch": 0.011807019719707304, "grad_norm": 3.024411916732788, "learning_rate": 7.624999999999999e-07, "loss": -0.0171, "num_tokens": 2627928.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6516.0, "completions/max_terminated_length": 6516.0, "completions/mean_length": 6088.5, "completions/mean_terminated_length": 6088.5, "completions/min_length": 5661.0, "completions/min_terminated_length": 5661.0, "epoch": 0.011831824382984002, "grad_norm": 0.0, "learning_rate": 7.62e-07, "loss": 0.0, "num_tokens": 2640999.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 808.0, "completions/max_terminated_length": 808.0, "completions/mean_length": 741.0, "completions/mean_terminated_length": 741.0, "completions/min_length": 674.0, "completions/min_terminated_length": 674.0, "epoch": 0.011856629046260697, "grad_norm": 0.0, "learning_rate": 7.614999999999999e-07, "loss": 0.0, "num_tokens": 2643297.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 478 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5227.0, "completions/mean_length": 6709.5, "completions/mean_terminated_length": 5227.0, "completions/min_length": 5227.0, "completions/min_terminated_length": 5227.0, "epoch": 0.011881433709537393, "grad_norm": 4.015349388122559, "learning_rate": 7.61e-07, "loss": -0.707, "num_tokens": 2649420.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 479 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01190623837281409, "grad_norm": 0.0, "learning_rate": 7.605e-07, "loss": 0.0, "num_tokens": 2650642.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 480 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.011931043036090785, "grad_norm": 0.0, "learning_rate": 7.599999999999999e-07, "loss": 0.0, "num_tokens": 2651634.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 481 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8129.0, "completions/max_terminated_length": 8129.0, "completions/mean_length": 6863.0, "completions/mean_terminated_length": 6863.0, "completions/min_length": 5597.0, "completions/min_terminated_length": 5597.0, "epoch": 0.011955847699367481, "grad_norm": 0.0, "learning_rate": 7.594999999999999e-07, "loss": 0.0, "num_tokens": 2666366.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 482 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2673.0, "completions/mean_length": 5432.5, "completions/mean_terminated_length": 2673.0, "completions/min_length": 2673.0, "completions/min_terminated_length": 2673.0, "epoch": 0.011980652362644176, "grad_norm": 0.0, "learning_rate": 7.59e-07, "loss": 0.0, "num_tokens": 2669949.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4312.0, "completions/mean_length": 6252.0, "completions/mean_terminated_length": 4312.0, "completions/min_length": 4312.0, "completions/min_terminated_length": 4312.0, "epoch": 0.012005457025920874, "grad_norm": 3.697305679321289, "learning_rate": 7.584999999999999e-07, "loss": -0.707, "num_tokens": 2675211.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2499.0, "completions/max_terminated_length": 2499.0, "completions/mean_length": 2312.0, "completions/mean_terminated_length": 2312.0, "completions/min_length": 2125.0, "completions/min_terminated_length": 2125.0, "epoch": 0.01203026168919757, "grad_norm": 0.0, "learning_rate": 7.58e-07, "loss": 0.0, "num_tokens": 2680791.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 485 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012055066352474265, "grad_norm": 0.0, "learning_rate": 7.575e-07, "loss": 0.0, "num_tokens": 2681693.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6759.0, "completions/max_terminated_length": 6759.0, "completions/mean_length": 5539.0, "completions/mean_terminated_length": 5539.0, "completions/min_length": 4319.0, "completions/min_terminated_length": 4319.0, "epoch": 0.012079871015750962, "grad_norm": 0.0, "learning_rate": 7.57e-07, "loss": 0.0, "num_tokens": 2693629.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 487 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3561.0, "completions/max_terminated_length": 3561.0, "completions/mean_length": 2452.5, "completions/mean_terminated_length": 2452.5, "completions/min_length": 1344.0, "completions/min_terminated_length": 1344.0, "epoch": 0.012104675679027658, "grad_norm": 0.0, "learning_rate": 7.564999999999999e-07, "loss": 0.0, "num_tokens": 2699380.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 488 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1938.0, "completions/max_terminated_length": 1938.0, "completions/mean_length": 1841.5, "completions/mean_terminated_length": 1841.5, "completions/min_length": 1745.0, "completions/min_terminated_length": 1745.0, "epoch": 0.012129480342304353, "grad_norm": 0.0, "learning_rate": 7.559999999999999e-07, "loss": 0.0, "num_tokens": 2703949.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 489 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7527.0, "completions/max_terminated_length": 7527.0, "completions/mean_length": 7353.0, "completions/mean_terminated_length": 7353.0, "completions/min_length": 7179.0, "completions/min_terminated_length": 7179.0, "epoch": 0.012154285005581049, "grad_norm": 0.0, "learning_rate": 7.554999999999999e-07, "loss": 0.0, "num_tokens": 2719533.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4402.0, "completions/max_terminated_length": 4402.0, "completions/mean_length": 4339.0, "completions/mean_terminated_length": 4339.0, "completions/min_length": 4276.0, "completions/min_terminated_length": 4276.0, "epoch": 0.012179089668857746, "grad_norm": 0.0, "learning_rate": 7.55e-07, "loss": 0.0, "num_tokens": 2729393.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5878.0, "completions/max_terminated_length": 5878.0, "completions/mean_length": 4598.5, "completions/mean_terminated_length": 4598.5, "completions/min_length": 3319.0, "completions/min_terminated_length": 3319.0, "epoch": 0.012203894332134441, "grad_norm": 0.0, "learning_rate": 7.544999999999999e-07, "loss": 0.0, "num_tokens": 2739578.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 492 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012228698995411137, "grad_norm": 0.0, "learning_rate": 7.54e-07, "loss": 0.0, "num_tokens": 2740618.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1595.0, "completions/max_terminated_length": 1595.0, "completions/mean_length": 1502.0, "completions/mean_terminated_length": 1502.0, "completions/min_length": 1409.0, "completions/min_terminated_length": 1409.0, "epoch": 0.012253503658687834, "grad_norm": 0.0, "learning_rate": 7.535e-07, "loss": 0.0, "num_tokens": 2744404.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 494 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01227830832196453, "grad_norm": 0.0, "learning_rate": 7.529999999999999e-07, "loss": 0.0, "num_tokens": 2745350.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1901.0, "completions/mean_length": 5046.5, "completions/mean_terminated_length": 1901.0, "completions/min_length": 1901.0, "completions/min_terminated_length": 1901.0, "epoch": 0.012303112985241225, "grad_norm": 0.0, "learning_rate": 7.524999999999999e-07, "loss": 0.0, "num_tokens": 2748159.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 496 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01232791764851792, "grad_norm": 0.0, "learning_rate": 7.52e-07, "loss": 0.0, "num_tokens": 2749003.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2226.0, "completions/max_terminated_length": 2226.0, "completions/mean_length": 1718.5, "completions/mean_terminated_length": 1718.5, "completions/min_length": 1211.0, "completions/min_terminated_length": 1211.0, "epoch": 0.012352722311794618, "grad_norm": 4.434837818145752, "learning_rate": 7.514999999999999e-07, "loss": 0.2088, "num_tokens": 2753280.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 498 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012377526975071313, "grad_norm": 0.0, "learning_rate": 7.51e-07, "loss": 0.0, "num_tokens": 2754114.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 499 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2967.0, "completions/max_terminated_length": 2967.0, "completions/mean_length": 2870.5, "completions/mean_terminated_length": 2870.5, "completions/min_length": 2774.0, "completions/min_terminated_length": 2774.0, "epoch": 0.012402331638348009, "grad_norm": 0.0, "learning_rate": 7.505e-07, "loss": 0.0, "num_tokens": 2760691.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 500 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012427136301624706, "grad_norm": 0.0, "learning_rate": 7.5e-07, "loss": 0.0, "num_tokens": 2761685.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 501 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012451940964901402, "grad_norm": 0.0, "learning_rate": 7.495e-07, "loss": 0.0, "num_tokens": 2762647.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 502 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6247.0, "completions/max_terminated_length": 6247.0, "completions/mean_length": 4363.5, "completions/mean_terminated_length": 4363.5, "completions/min_length": 2480.0, "completions/min_terminated_length": 2480.0, "epoch": 0.012476745628178097, "grad_norm": 3.3324625492095947, "learning_rate": 7.489999999999999e-07, "loss": -0.3052, "num_tokens": 2772246.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 503 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7280.0, "completions/mean_length": 7736.0, "completions/mean_terminated_length": 7280.0, "completions/min_length": 7280.0, "completions/min_terminated_length": 7280.0, "epoch": 0.012501550291454793, "grad_norm": 2.7449092864990234, "learning_rate": 7.485e-07, "loss": -0.707, "num_tokens": 2780516.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1658.0, "completions/max_terminated_length": 1658.0, "completions/mean_length": 1263.5, "completions/mean_terminated_length": 1263.5, "completions/min_length": 869.0, "completions/min_terminated_length": 869.0, "epoch": 0.01252635495473149, "grad_norm": 0.0, "learning_rate": 7.48e-07, "loss": 0.0, "num_tokens": 2783863.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 975.0, "completions/max_terminated_length": 975.0, "completions/mean_length": 849.0, "completions/mean_terminated_length": 849.0, "completions/min_length": 723.0, "completions/min_terminated_length": 723.0, "epoch": 0.012551159618008186, "grad_norm": 0.0, "learning_rate": 7.475e-07, "loss": 0.0, "num_tokens": 2786389.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5448.0, "completions/max_terminated_length": 5448.0, "completions/mean_length": 5047.0, "completions/mean_terminated_length": 5047.0, "completions/min_length": 4646.0, "completions/min_terminated_length": 4646.0, "epoch": 0.012575964281284881, "grad_norm": 2.7631521224975586, "learning_rate": 7.47e-07, "loss": 0.0562, "num_tokens": 2797417.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 507 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2364.0, "completions/max_terminated_length": 2364.0, "completions/mean_length": 1824.5, "completions/mean_terminated_length": 1824.5, "completions/min_length": 1285.0, "completions/min_terminated_length": 1285.0, "epoch": 0.012600768944561578, "grad_norm": 0.0, "learning_rate": 7.465e-07, "loss": 0.0, "num_tokens": 2801882.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5636.0, "completions/mean_length": 6914.0, "completions/mean_terminated_length": 5636.0, "completions/min_length": 5636.0, "completions/min_terminated_length": 5636.0, "epoch": 0.012625573607838274, "grad_norm": 0.0, "learning_rate": 7.459999999999999e-07, "loss": 0.0, "num_tokens": 2808348.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3697.0, "completions/max_terminated_length": 3697.0, "completions/mean_length": 3646.5, "completions/mean_terminated_length": 3646.5, "completions/min_length": 3596.0, "completions/min_terminated_length": 3596.0, "epoch": 0.01265037827111497, "grad_norm": 0.0, "learning_rate": 7.455e-07, "loss": 0.0, "num_tokens": 2816577.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 510 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1616.0, "completions/max_terminated_length": 1616.0, "completions/mean_length": 949.5, "completions/mean_terminated_length": 949.5, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "epoch": 0.012675182934391665, "grad_norm": 0.0, "learning_rate": 7.45e-07, "loss": 0.0, "num_tokens": 2819296.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 511 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012699987597668362, "grad_norm": 0.0, "learning_rate": 7.445e-07, "loss": 0.0, "num_tokens": 2820170.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 512 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012724792260945058, "grad_norm": 0.0, "learning_rate": 7.44e-07, "loss": 0.0, "num_tokens": 2821278.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 513 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012749596924221753, "grad_norm": 0.0, "learning_rate": 7.435000000000001e-07, "loss": 0.0, "num_tokens": 2822174.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5188.0, "completions/max_terminated_length": 5188.0, "completions/mean_length": 4001.0, "completions/mean_terminated_length": 4001.0, "completions/min_length": 2814.0, "completions/min_terminated_length": 2814.0, "epoch": 0.01277440158749845, "grad_norm": 0.0, "learning_rate": 7.429999999999999e-07, "loss": 0.0, "num_tokens": 2831048.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 515 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012799206250775146, "grad_norm": 0.0, "learning_rate": 7.425e-07, "loss": 0.0, "num_tokens": 2831932.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2206.0, "completions/max_terminated_length": 2206.0, "completions/mean_length": 1917.0, "completions/mean_terminated_length": 1917.0, "completions/min_length": 1628.0, "completions/min_terminated_length": 1628.0, "epoch": 0.012824010914051841, "grad_norm": 0.0, "learning_rate": 7.42e-07, "loss": 0.0, "num_tokens": 2836658.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7837.0, "completions/mean_length": 8014.5, "completions/mean_terminated_length": 7837.0, "completions/min_length": 7837.0, "completions/min_terminated_length": 7837.0, "epoch": 0.012848815577328537, "grad_norm": 0.0, "learning_rate": 7.415e-07, "loss": 0.0, "num_tokens": 2845797.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3907.0, "completions/max_terminated_length": 3907.0, "completions/mean_length": 3199.5, "completions/mean_terminated_length": 3199.5, "completions/min_length": 2492.0, "completions/min_terminated_length": 2492.0, "epoch": 0.012873620240605234, "grad_norm": 3.7661781311035156, "learning_rate": 7.41e-07, "loss": -0.1563, "num_tokens": 2853084.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 519 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1062.0, "completions/max_terminated_length": 1062.0, "completions/mean_length": 928.5, "completions/mean_terminated_length": 928.5, "completions/min_length": 795.0, "completions/min_terminated_length": 795.0, "epoch": 0.01289842490388193, "grad_norm": 0.0, "learning_rate": 7.405e-07, "loss": 0.0, "num_tokens": 2855823.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 520 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.012923229567158625, "grad_norm": 0.0, "learning_rate": 7.4e-07, "loss": 0.0, "num_tokens": 2856897.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 521 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1850.0, "completions/max_terminated_length": 1850.0, "completions/mean_length": 1692.0, "completions/mean_terminated_length": 1692.0, "completions/min_length": 1534.0, "completions/min_terminated_length": 1534.0, "epoch": 0.012948034230435322, "grad_norm": 0.0, "learning_rate": 7.395e-07, "loss": 0.0, "num_tokens": 2861179.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 522 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6672.0, "completions/max_terminated_length": 6672.0, "completions/mean_length": 4169.5, "completions/mean_terminated_length": 4169.5, "completions/min_length": 1667.0, "completions/min_terminated_length": 1667.0, "epoch": 0.012972838893712018, "grad_norm": 0.0, "learning_rate": 7.389999999999999e-07, "loss": 0.0, "num_tokens": 2870366.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2097.0, "completions/max_terminated_length": 2097.0, "completions/mean_length": 1749.0, "completions/mean_terminated_length": 1749.0, "completions/min_length": 1401.0, "completions/min_terminated_length": 1401.0, "epoch": 0.012997643556988714, "grad_norm": 0.0, "learning_rate": 7.385e-07, "loss": 0.0, "num_tokens": 2874702.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 524 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013022448220265409, "grad_norm": 0.0, "learning_rate": 7.38e-07, "loss": 0.0, "num_tokens": 2875614.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7846.0, "completions/mean_length": 8019.0, "completions/mean_terminated_length": 7846.0, "completions/min_length": 7846.0, "completions/min_terminated_length": 7846.0, "epoch": 0.013047252883542106, "grad_norm": 2.947042942047119, "learning_rate": 7.375e-07, "loss": -0.707, "num_tokens": 2884460.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 526 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013072057546818802, "grad_norm": 0.0, "learning_rate": 7.37e-07, "loss": 0.0, "num_tokens": 2885352.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 527 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3922.0, "completions/mean_length": 6057.0, "completions/mean_terminated_length": 3922.0, "completions/min_length": 3922.0, "completions/min_terminated_length": 3922.0, "epoch": 0.013096862210095497, "grad_norm": 4.492016315460205, "learning_rate": 7.365e-07, "loss": -0.707, "num_tokens": 2890170.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7607.0, "completions/max_terminated_length": 7607.0, "completions/mean_length": 6609.5, "completions/mean_terminated_length": 6609.5, "completions/min_length": 5612.0, "completions/min_terminated_length": 5612.0, "epoch": 0.013121666873372195, "grad_norm": 0.0, "learning_rate": 7.359999999999999e-07, "loss": 0.0, "num_tokens": 2904295.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1997.0, "completions/max_terminated_length": 1997.0, "completions/mean_length": 1770.5, "completions/mean_terminated_length": 1770.5, "completions/min_length": 1544.0, "completions/min_terminated_length": 1544.0, "epoch": 0.01314647153664889, "grad_norm": 0.0, "learning_rate": 7.355e-07, "loss": 0.0, "num_tokens": 2908668.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 530 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7671.0, "completions/max_terminated_length": 7671.0, "completions/mean_length": 6280.0, "completions/mean_terminated_length": 6280.0, "completions/min_length": 4889.0, "completions/min_terminated_length": 4889.0, "epoch": 0.013171276199925586, "grad_norm": 0.0, "learning_rate": 7.35e-07, "loss": 0.0, "num_tokens": 2922178.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1064.0, "completions/max_terminated_length": 1064.0, "completions/mean_length": 871.5, "completions/mean_terminated_length": 871.5, "completions/min_length": 679.0, "completions/min_terminated_length": 679.0, "epoch": 0.013196080863202283, "grad_norm": 0.0, "learning_rate": 7.345e-07, "loss": 0.0, "num_tokens": 2924753.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 532 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013220885526478978, "grad_norm": 0.0, "learning_rate": 7.34e-07, "loss": 0.0, "num_tokens": 2925587.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 533 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2271.0, "completions/max_terminated_length": 2271.0, "completions/mean_length": 2234.0, "completions/mean_terminated_length": 2234.0, "completions/min_length": 2197.0, "completions/min_terminated_length": 2197.0, "epoch": 0.013245690189755674, "grad_norm": 0.0, "learning_rate": 7.335e-07, "loss": 0.0, "num_tokens": 2930881.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1765.0, "completions/max_terminated_length": 1765.0, "completions/mean_length": 1381.0, "completions/mean_terminated_length": 1381.0, "completions/min_length": 997.0, "completions/min_terminated_length": 997.0, "epoch": 0.01327049485303237, "grad_norm": 0.0, "learning_rate": 7.329999999999999e-07, "loss": 0.0, "num_tokens": 2934497.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 535 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013295299516309067, "grad_norm": 0.0, "learning_rate": 7.325e-07, "loss": 0.0, "num_tokens": 2935367.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3588.0, "completions/mean_length": 5890.0, "completions/mean_terminated_length": 3588.0, "completions/min_length": 3588.0, "completions/min_terminated_length": 3588.0, "epoch": 0.013320104179585762, "grad_norm": 0.0, "learning_rate": 7.319999999999999e-07, "loss": 0.0, "num_tokens": 2940021.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 537 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1009.0, "completions/max_terminated_length": 1009.0, "completions/mean_length": 845.5, "completions/mean_terminated_length": 845.5, "completions/min_length": 682.0, "completions/min_terminated_length": 682.0, "epoch": 0.013344908842862458, "grad_norm": 0.0, "learning_rate": 7.315e-07, "loss": 0.0, "num_tokens": 2942552.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2491.0, "completions/max_terminated_length": 2491.0, "completions/mean_length": 1773.0, "completions/mean_terminated_length": 1773.0, "completions/min_length": 1055.0, "completions/min_terminated_length": 1055.0, "epoch": 0.013369713506139155, "grad_norm": 0.0, "learning_rate": 7.31e-07, "loss": 0.0, "num_tokens": 2946936.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 539 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 666.0, "completions/max_terminated_length": 666.0, "completions/mean_length": 637.0, "completions/mean_terminated_length": 637.0, "completions/min_length": 608.0, "completions/min_terminated_length": 608.0, "epoch": 0.01339451816941585, "grad_norm": 0.0, "learning_rate": 7.305e-07, "loss": 0.0, "num_tokens": 2949068.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4947.0, "completions/mean_length": 6569.5, "completions/mean_terminated_length": 4947.0, "completions/min_length": 4947.0, "completions/min_terminated_length": 4947.0, "epoch": 0.013419322832692546, "grad_norm": 3.372856378555298, "learning_rate": 7.3e-07, "loss": -0.707, "num_tokens": 2954837.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 541 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013444127495969242, "grad_norm": 0.0, "learning_rate": 7.295e-07, "loss": 0.0, "num_tokens": 2955797.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 542 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013468932159245939, "grad_norm": 0.0, "learning_rate": 7.289999999999999e-07, "loss": 0.0, "num_tokens": 2956655.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 543 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013493736822522634, "grad_norm": 0.0, "learning_rate": 7.285e-07, "loss": 0.0, "num_tokens": 2957583.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 544 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01351854148579933, "grad_norm": 0.0, "learning_rate": 7.28e-07, "loss": 0.0, "num_tokens": 2958489.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8111.0, "completions/max_terminated_length": 8111.0, "completions/mean_length": 5797.0, "completions/mean_terminated_length": 5797.0, "completions/min_length": 3483.0, "completions/min_terminated_length": 3483.0, "epoch": 0.013543346149076027, "grad_norm": 0.0, "learning_rate": 7.275e-07, "loss": 0.0, "num_tokens": 2970937.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 546 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013568150812352723, "grad_norm": 0.0, "learning_rate": 7.27e-07, "loss": 0.0, "num_tokens": 2971873.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5572.0, "completions/mean_length": 6882.0, "completions/mean_terminated_length": 5572.0, "completions/min_length": 5572.0, "completions/min_terminated_length": 5572.0, "epoch": 0.013592955475629418, "grad_norm": 3.947927713394165, "learning_rate": 7.265000000000001e-07, "loss": -0.707, "num_tokens": 2978269.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6844.0, "completions/max_terminated_length": 6844.0, "completions/mean_length": 6480.0, "completions/mean_terminated_length": 6480.0, "completions/min_length": 6116.0, "completions/min_terminated_length": 6116.0, "epoch": 0.013617760138906114, "grad_norm": 0.0, "learning_rate": 7.259999999999999e-07, "loss": 0.0, "num_tokens": 2992143.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4893.0, "completions/mean_length": 6542.5, "completions/mean_terminated_length": 4893.0, "completions/min_length": 4893.0, "completions/min_terminated_length": 4893.0, "epoch": 0.013642564802182811, "grad_norm": 4.513481140136719, "learning_rate": 7.255e-07, "loss": -0.707, "num_tokens": 2998386.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 550 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.013667369465459506, "grad_norm": 0.0, "learning_rate": 7.249999999999999e-07, "loss": 0.0, "num_tokens": 2999260.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4281.0, "completions/max_terminated_length": 4281.0, "completions/mean_length": 4177.0, "completions/mean_terminated_length": 4177.0, "completions/min_length": 4073.0, "completions/min_terminated_length": 4073.0, "epoch": 0.013692174128736202, "grad_norm": 0.0, "learning_rate": 7.245e-07, "loss": 0.0, "num_tokens": 3008576.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 552 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0137169787920129, "grad_norm": 0.0, "learning_rate": 7.24e-07, "loss": 0.0, "num_tokens": 3009450.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 987.0, "completions/max_terminated_length": 987.0, "completions/mean_length": 764.0, "completions/mean_terminated_length": 764.0, "completions/min_length": 541.0, "completions/min_terminated_length": 541.0, "epoch": 0.013741783455289595, "grad_norm": 0.0, "learning_rate": 7.235e-07, "loss": 0.0, "num_tokens": 3011766.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4485.0, "completions/mean_length": 6338.5, "completions/mean_terminated_length": 4485.0, "completions/min_length": 4485.0, "completions/min_terminated_length": 4485.0, "epoch": 0.01376658811856629, "grad_norm": 3.5541749000549316, "learning_rate": 7.229999999999999e-07, "loss": -0.707, "num_tokens": 3017121.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7313.0, "completions/max_terminated_length": 7313.0, "completions/mean_length": 5947.5, "completions/mean_terminated_length": 5947.5, "completions/min_length": 4582.0, "completions/min_terminated_length": 4582.0, "epoch": 0.013791392781842986, "grad_norm": 3.0289061069488525, "learning_rate": 7.225e-07, "loss": 0.1623, "num_tokens": 3029878.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7349.0, "completions/mean_length": 7770.5, "completions/mean_terminated_length": 7349.0, "completions/min_length": 7349.0, "completions/min_terminated_length": 7349.0, "epoch": 0.013816197445119683, "grad_norm": 4.430403232574463, "learning_rate": 7.219999999999999e-07, "loss": -0.707, "num_tokens": 3038103.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8020.0, "completions/max_terminated_length": 8020.0, "completions/mean_length": 5583.0, "completions/mean_terminated_length": 5583.0, "completions/min_length": 3146.0, "completions/min_terminated_length": 3146.0, "epoch": 0.013841002108396378, "grad_norm": 0.0, "learning_rate": 7.215e-07, "loss": 0.0, "num_tokens": 3050177.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 614.0, "completions/max_terminated_length": 614.0, "completions/mean_length": 561.0, "completions/mean_terminated_length": 561.0, "completions/min_length": 508.0, "completions/min_terminated_length": 508.0, "epoch": 0.013865806771673074, "grad_norm": 0.0, "learning_rate": 7.21e-07, "loss": 0.0, "num_tokens": 3052135.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 559 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3609.0, "completions/max_terminated_length": 3609.0, "completions/mean_length": 3165.0, "completions/mean_terminated_length": 3165.0, "completions/min_length": 2721.0, "completions/min_terminated_length": 2721.0, "epoch": 0.013890611434949771, "grad_norm": 0.0, "learning_rate": 7.205e-07, "loss": 0.0, "num_tokens": 3059371.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5694.0, "completions/max_terminated_length": 5694.0, "completions/mean_length": 5317.0, "completions/mean_terminated_length": 5317.0, "completions/min_length": 4940.0, "completions/min_terminated_length": 4940.0, "epoch": 0.013915416098226467, "grad_norm": 0.0, "learning_rate": 7.2e-07, "loss": 0.0, "num_tokens": 3071093.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 561 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2964.0, "completions/max_terminated_length": 2964.0, "completions/mean_length": 2063.5, "completions/mean_terminated_length": 2063.5, "completions/min_length": 1163.0, "completions/min_terminated_length": 1163.0, "epoch": 0.013940220761503162, "grad_norm": 0.0, "learning_rate": 7.195e-07, "loss": 0.0, "num_tokens": 3076208.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 562 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5621.0, "completions/max_terminated_length": 5621.0, "completions/mean_length": 5226.0, "completions/mean_terminated_length": 5226.0, "completions/min_length": 4831.0, "completions/min_terminated_length": 4831.0, "epoch": 0.013965025424779858, "grad_norm": 0.0, "learning_rate": 7.189999999999999e-07, "loss": 0.0, "num_tokens": 3087520.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 960.0, "completions/max_terminated_length": 960.0, "completions/mean_length": 720.0, "completions/mean_terminated_length": 720.0, "completions/min_length": 480.0, "completions/min_terminated_length": 480.0, "epoch": 0.013989830088056555, "grad_norm": 0.0, "learning_rate": 7.185e-07, "loss": 0.0, "num_tokens": 3089814.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1721.0, "completions/max_terminated_length": 1721.0, "completions/mean_length": 1583.5, "completions/mean_terminated_length": 1583.5, "completions/min_length": 1446.0, "completions/min_terminated_length": 1446.0, "epoch": 0.01401463475133325, "grad_norm": 0.0, "learning_rate": 7.179999999999999e-07, "loss": 0.0, "num_tokens": 3093849.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2890.0, "completions/max_terminated_length": 2890.0, "completions/mean_length": 2310.0, "completions/mean_terminated_length": 2310.0, "completions/min_length": 1730.0, "completions/min_terminated_length": 1730.0, "epoch": 0.014039439414609946, "grad_norm": 0.0, "learning_rate": 7.175e-07, "loss": 0.0, "num_tokens": 3099315.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 566 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1887.0, "completions/max_terminated_length": 1887.0, "completions/mean_length": 1800.0, "completions/mean_terminated_length": 1800.0, "completions/min_length": 1713.0, "completions/min_terminated_length": 1713.0, "epoch": 0.014064244077886643, "grad_norm": 0.0, "learning_rate": 7.17e-07, "loss": 0.0, "num_tokens": 3103721.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4398.0, "completions/max_terminated_length": 4398.0, "completions/mean_length": 3671.5, "completions/mean_terminated_length": 3671.5, "completions/min_length": 2945.0, "completions/min_terminated_length": 2945.0, "epoch": 0.014089048741163339, "grad_norm": 0.0, "learning_rate": 7.165e-07, "loss": 0.0, "num_tokens": 3111940.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 568 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6826.0, "completions/mean_length": 7509.0, "completions/mean_terminated_length": 6826.0, "completions/min_length": 6826.0, "completions/min_terminated_length": 6826.0, "epoch": 0.014113853404440034, "grad_norm": 0.0, "learning_rate": 7.159999999999999e-07, "loss": 0.0, "num_tokens": 3119654.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 569 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01413865806771673, "grad_norm": 0.0, "learning_rate": 7.155e-07, "loss": 0.0, "num_tokens": 3120580.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5214.0, "completions/max_terminated_length": 5214.0, "completions/mean_length": 3641.0, "completions/mean_terminated_length": 3641.0, "completions/min_length": 2068.0, "completions/min_terminated_length": 2068.0, "epoch": 0.014163462730993427, "grad_norm": 0.0, "learning_rate": 7.149999999999999e-07, "loss": 0.0, "num_tokens": 3128870.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 571 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014188267394270123, "grad_norm": 0.0, "learning_rate": 7.145e-07, "loss": 0.0, "num_tokens": 3129898.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 572 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6799.0, "completions/mean_length": 7495.5, "completions/mean_terminated_length": 6799.0, "completions/min_length": 6799.0, "completions/min_terminated_length": 6799.0, "epoch": 0.014213072057546818, "grad_norm": 2.747023582458496, "learning_rate": 7.14e-07, "loss": -0.707, "num_tokens": 3137517.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7984.0, "completions/max_terminated_length": 7984.0, "completions/mean_length": 6646.0, "completions/mean_terminated_length": 6646.0, "completions/min_length": 5308.0, "completions/min_terminated_length": 5308.0, "epoch": 0.014237876720823515, "grad_norm": 0.0, "learning_rate": 7.135e-07, "loss": 0.0, "num_tokens": 3151721.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5678.0, "completions/max_terminated_length": 5678.0, "completions/mean_length": 5062.5, "completions/mean_terminated_length": 5062.5, "completions/min_length": 4447.0, "completions/min_terminated_length": 4447.0, "epoch": 0.014262681384100211, "grad_norm": 2.7427237033843994, "learning_rate": 7.129999999999999e-07, "loss": 0.086, "num_tokens": 3162874.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7542.0, "completions/mean_length": 7867.0, "completions/mean_terminated_length": 7542.0, "completions/min_length": 7542.0, "completions/min_terminated_length": 7542.0, "epoch": 0.014287486047376907, "grad_norm": 0.0, "learning_rate": 7.125e-07, "loss": 0.0, "num_tokens": 3171436.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 956.0, "completions/max_terminated_length": 956.0, "completions/mean_length": 950.0, "completions/mean_terminated_length": 950.0, "completions/min_length": 944.0, "completions/min_terminated_length": 944.0, "epoch": 0.014312290710653602, "grad_norm": 0.0, "learning_rate": 7.119999999999999e-07, "loss": 0.0, "num_tokens": 3174156.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 577 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0143370953739303, "grad_norm": 0.0, "learning_rate": 7.115e-07, "loss": 0.0, "num_tokens": 3175204.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 578 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014361900037206995, "grad_norm": 0.0, "learning_rate": 7.11e-07, "loss": 0.0, "num_tokens": 3176172.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 579 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3490.0, "completions/max_terminated_length": 3490.0, "completions/mean_length": 2912.0, "completions/mean_terminated_length": 2912.0, "completions/min_length": 2334.0, "completions/min_terminated_length": 2334.0, "epoch": 0.01438670470048369, "grad_norm": 3.2014315128326416, "learning_rate": 7.105e-07, "loss": 0.1403, "num_tokens": 3182838.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8142.0, "completions/max_terminated_length": 8142.0, "completions/mean_length": 5956.5, "completions/mean_terminated_length": 5956.5, "completions/min_length": 3771.0, "completions/min_terminated_length": 3771.0, "epoch": 0.014411509363760388, "grad_norm": 0.0, "learning_rate": 7.1e-07, "loss": 0.0, "num_tokens": 3195559.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7041.0, "completions/max_terminated_length": 7041.0, "completions/mean_length": 5010.0, "completions/mean_terminated_length": 5010.0, "completions/min_length": 2979.0, "completions/min_terminated_length": 2979.0, "epoch": 0.014436314027037083, "grad_norm": 0.0, "learning_rate": 7.094999999999999e-07, "loss": 0.0, "num_tokens": 3206503.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 582 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014461118690313779, "grad_norm": 0.0, "learning_rate": 7.089999999999999e-07, "loss": 0.0, "num_tokens": 3207521.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 583 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014485923353590476, "grad_norm": 0.0, "learning_rate": 7.085e-07, "loss": 0.0, "num_tokens": 3208723.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 584 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014510728016867171, "grad_norm": 0.0, "learning_rate": 7.079999999999999e-07, "loss": 0.0, "num_tokens": 3209737.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2367.0, "completions/mean_length": 5279.5, "completions/mean_terminated_length": 2367.0, "completions/min_length": 2367.0, "completions/min_terminated_length": 2367.0, "epoch": 0.014535532680143867, "grad_norm": 4.785103797912598, "learning_rate": 7.075e-07, "loss": -0.707, "num_tokens": 3213036.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2132.0, "completions/max_terminated_length": 2132.0, "completions/mean_length": 2113.0, "completions/mean_terminated_length": 2113.0, "completions/min_length": 2094.0, "completions/min_terminated_length": 2094.0, "epoch": 0.014560337343420562, "grad_norm": 0.0, "learning_rate": 7.07e-07, "loss": 0.0, "num_tokens": 3218606.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 587 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01458514200669726, "grad_norm": 0.0, "learning_rate": 7.065e-07, "loss": 0.0, "num_tokens": 3219418.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 588 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014609946669973955, "grad_norm": 0.0, "learning_rate": 7.059999999999999e-07, "loss": 0.0, "num_tokens": 3220386.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2376.0, "completions/max_terminated_length": 2376.0, "completions/mean_length": 1880.5, "completions/mean_terminated_length": 1880.5, "completions/min_length": 1385.0, "completions/min_terminated_length": 1385.0, "epoch": 0.01463475133325065, "grad_norm": 0.0, "learning_rate": 7.055e-07, "loss": 0.0, "num_tokens": 3224963.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 590 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014659555996527348, "grad_norm": 0.0, "learning_rate": 7.049999999999999e-07, "loss": 0.0, "num_tokens": 3225805.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 591 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7452.0, "completions/max_terminated_length": 7452.0, "completions/mean_length": 5660.0, "completions/mean_terminated_length": 5660.0, "completions/min_length": 3868.0, "completions/min_terminated_length": 3868.0, "epoch": 0.014684360659804043, "grad_norm": 0.0, "learning_rate": 7.045e-07, "loss": 0.0, "num_tokens": 3238101.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 592 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 563.0, "completions/max_terminated_length": 563.0, "completions/mean_length": 460.0, "completions/mean_terminated_length": 460.0, "completions/min_length": 357.0, "completions/min_terminated_length": 357.0, "epoch": 0.014709165323080739, "grad_norm": 0.0, "learning_rate": 7.04e-07, "loss": 0.0, "num_tokens": 3239809.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6682.0, "completions/mean_length": 7437.0, "completions/mean_terminated_length": 6682.0, "completions/min_length": 6682.0, "completions/min_terminated_length": 6682.0, "epoch": 0.014733969986357435, "grad_norm": 0.0, "learning_rate": 7.035e-07, "loss": 0.0, "num_tokens": 3247511.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5844.0, "completions/max_terminated_length": 5844.0, "completions/mean_length": 5286.0, "completions/mean_terminated_length": 5286.0, "completions/min_length": 4728.0, "completions/min_terminated_length": 4728.0, "epoch": 0.014758774649634132, "grad_norm": 0.0, "learning_rate": 7.029999999999999e-07, "loss": 0.0, "num_tokens": 3258965.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6802.0, "completions/mean_length": 7497.0, "completions/mean_terminated_length": 6802.0, "completions/min_length": 6802.0, "completions/min_terminated_length": 6802.0, "epoch": 0.014783579312910827, "grad_norm": 3.8531720638275146, "learning_rate": 7.024999999999999e-07, "loss": -0.707, "num_tokens": 3266677.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4532.0, "completions/mean_length": 6362.0, "completions/mean_terminated_length": 4532.0, "completions/min_length": 4532.0, "completions/min_terminated_length": 4532.0, "epoch": 0.014808383976187523, "grad_norm": 3.474674701690674, "learning_rate": 7.019999999999999e-07, "loss": -0.707, "num_tokens": 3272063.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8104.0, "completions/mean_length": 8148.0, "completions/mean_terminated_length": 8104.0, "completions/min_length": 8104.0, "completions/min_terminated_length": 8104.0, "epoch": 0.01483318863946422, "grad_norm": 2.8230373859405518, "learning_rate": 7.015e-07, "loss": -0.707, "num_tokens": 3281041.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4799.0, "completions/max_terminated_length": 4799.0, "completions/mean_length": 4567.0, "completions/mean_terminated_length": 4567.0, "completions/min_length": 4335.0, "completions/min_terminated_length": 4335.0, "epoch": 0.014857993302740916, "grad_norm": 0.0, "learning_rate": 7.009999999999999e-07, "loss": 0.0, "num_tokens": 3291217.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2058.0, "completions/max_terminated_length": 2058.0, "completions/mean_length": 2054.0, "completions/mean_terminated_length": 2054.0, "completions/min_length": 2050.0, "completions/min_terminated_length": 2050.0, "epoch": 0.014882797966017611, "grad_norm": 0.0, "learning_rate": 7.005e-07, "loss": 0.0, "num_tokens": 3296225.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 600 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014907602629294307, "grad_norm": 0.0, "learning_rate": 7e-07, "loss": 0.0, "num_tokens": 3297097.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 601 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.014932407292571004, "grad_norm": 0.0, "learning_rate": 6.994999999999999e-07, "loss": 0.0, "num_tokens": 3298199.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 602 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0149572119558477, "grad_norm": 0.0, "learning_rate": 6.989999999999999e-07, "loss": 0.0, "num_tokens": 3299137.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 603 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7894.0, "completions/mean_length": 8043.0, "completions/mean_terminated_length": 7894.0, "completions/min_length": 7894.0, "completions/min_terminated_length": 7894.0, "epoch": 0.014982016619124395, "grad_norm": 0.0, "learning_rate": 6.985e-07, "loss": 0.0, "num_tokens": 3308025.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6385.0, "completions/max_terminated_length": 6385.0, "completions/mean_length": 5469.0, "completions/mean_terminated_length": 5469.0, "completions/min_length": 4553.0, "completions/min_terminated_length": 4553.0, "epoch": 0.015006821282401092, "grad_norm": 0.0, "learning_rate": 6.979999999999999e-07, "loss": 0.0, "num_tokens": 3319871.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 605 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015031625945677788, "grad_norm": 0.0, "learning_rate": 6.975e-07, "loss": 0.0, "num_tokens": 3320809.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 606 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015056430608954483, "grad_norm": 0.0, "learning_rate": 6.97e-07, "loss": 0.0, "num_tokens": 3321801.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 607 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015081235272231179, "grad_norm": 0.0, "learning_rate": 6.965e-07, "loss": 0.0, "num_tokens": 3322857.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3257.0, "completions/max_terminated_length": 3257.0, "completions/mean_length": 3158.5, "completions/mean_terminated_length": 3158.5, "completions/min_length": 3060.0, "completions/min_terminated_length": 3060.0, "epoch": 0.015106039935507876, "grad_norm": 4.024023056030273, "learning_rate": 6.959999999999999e-07, "loss": 0.022, "num_tokens": 3330056.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1602.0, "completions/max_terminated_length": 1602.0, "completions/mean_length": 1220.0, "completions/mean_terminated_length": 1220.0, "completions/min_length": 838.0, "completions/min_terminated_length": 838.0, "epoch": 0.015130844598784571, "grad_norm": 0.0, "learning_rate": 6.955e-07, "loss": 0.0, "num_tokens": 3333304.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 610 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1947.0, "completions/max_terminated_length": 1947.0, "completions/mean_length": 1461.5, "completions/mean_terminated_length": 1461.5, "completions/min_length": 976.0, "completions/min_terminated_length": 976.0, "epoch": 0.015155649262061267, "grad_norm": 0.0, "learning_rate": 6.949999999999999e-07, "loss": 0.0, "num_tokens": 3337031.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 611 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015180453925337964, "grad_norm": 0.0, "learning_rate": 6.945e-07, "loss": 0.0, "num_tokens": 3337845.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5428.0, "completions/max_terminated_length": 5428.0, "completions/mean_length": 4555.5, "completions/mean_terminated_length": 4555.5, "completions/min_length": 3683.0, "completions/min_terminated_length": 3683.0, "epoch": 0.01520525858861466, "grad_norm": 0.0, "learning_rate": 6.939999999999999e-07, "loss": 0.0, "num_tokens": 3347868.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 613 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015230063251891355, "grad_norm": 0.0, "learning_rate": 6.935e-07, "loss": 0.0, "num_tokens": 3348808.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6030.0, "completions/max_terminated_length": 6030.0, "completions/mean_length": 4350.5, "completions/mean_terminated_length": 4350.5, "completions/min_length": 2671.0, "completions/min_terminated_length": 2671.0, "epoch": 0.01525486791516805, "grad_norm": 0.0, "learning_rate": 6.929999999999999e-07, "loss": 0.0, "num_tokens": 3358593.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 615 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015279672578444748, "grad_norm": 0.0, "learning_rate": 6.924999999999999e-07, "loss": 0.0, "num_tokens": 3359519.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1617.0, "completions/max_terminated_length": 1617.0, "completions/mean_length": 1503.0, "completions/mean_terminated_length": 1503.0, "completions/min_length": 1389.0, "completions/min_terminated_length": 1389.0, "epoch": 0.015304477241721444, "grad_norm": 0.0, "learning_rate": 6.919999999999999e-07, "loss": 0.0, "num_tokens": 3363449.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 617 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 4278.0, "completions/mean_terminated_length": 364.0, "completions/min_length": 364.0, "completions/min_terminated_length": 364.0, "epoch": 0.015329281904998139, "grad_norm": 12.488121032714844, "learning_rate": 6.915e-07, "loss": -0.707, "num_tokens": 3364613.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 618 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7715.0, "completions/mean_length": 7953.5, "completions/mean_terminated_length": 7715.0, "completions/min_length": 7715.0, "completions/min_terminated_length": 7715.0, "epoch": 0.015354086568274836, "grad_norm": 3.3416998386383057, "learning_rate": 6.909999999999999e-07, "loss": -0.707, "num_tokens": 3373326.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 619 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8057.0, "completions/max_terminated_length": 8057.0, "completions/mean_length": 7529.0, "completions/mean_terminated_length": 7529.0, "completions/min_length": 7001.0, "completions/min_terminated_length": 7001.0, "epoch": 0.015378891231551532, "grad_norm": 2.4530749320983887, "learning_rate": 6.905e-07, "loss": 0.0496, "num_tokens": 3389218.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 620 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4717.0, "completions/mean_length": 6454.5, "completions/mean_terminated_length": 4717.0, "completions/min_length": 4717.0, "completions/min_terminated_length": 4717.0, "epoch": 0.015403695894828227, "grad_norm": 0.0, "learning_rate": 6.9e-07, "loss": 0.0, "num_tokens": 3394929.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 621 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 888.0, "completions/max_terminated_length": 888.0, "completions/mean_length": 841.5, "completions/mean_terminated_length": 841.5, "completions/min_length": 795.0, "completions/min_terminated_length": 795.0, "epoch": 0.015428500558104923, "grad_norm": 0.0, "learning_rate": 6.894999999999999e-07, "loss": 0.0, "num_tokens": 3397442.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 622 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01545330522138162, "grad_norm": 0.0, "learning_rate": 6.889999999999999e-07, "loss": 0.0, "num_tokens": 3398426.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 623 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7543.0, "completions/mean_length": 7867.5, "completions/mean_terminated_length": 7543.0, "completions/min_length": 7543.0, "completions/min_terminated_length": 7543.0, "epoch": 0.015478109884658316, "grad_norm": 3.429466724395752, "learning_rate": 6.885e-07, "loss": -0.707, "num_tokens": 3407571.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6201.0, "completions/max_terminated_length": 6201.0, "completions/mean_length": 5958.5, "completions/mean_terminated_length": 5958.5, "completions/min_length": 5716.0, "completions/min_terminated_length": 5716.0, "epoch": 0.015502914547935011, "grad_norm": 0.0, "learning_rate": 6.879999999999999e-07, "loss": 0.0, "num_tokens": 3420482.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5535.0, "completions/max_terminated_length": 5535.0, "completions/mean_length": 5301.0, "completions/mean_terminated_length": 5301.0, "completions/min_length": 5067.0, "completions/min_terminated_length": 5067.0, "epoch": 0.015527719211211708, "grad_norm": 0.0, "learning_rate": 6.875e-07, "loss": 0.0, "num_tokens": 3432182.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5318.0, "completions/max_terminated_length": 5318.0, "completions/mean_length": 4763.5, "completions/mean_terminated_length": 4763.5, "completions/min_length": 4209.0, "completions/min_terminated_length": 4209.0, "epoch": 0.015552523874488404, "grad_norm": 0.0, "learning_rate": 6.87e-07, "loss": 0.0, "num_tokens": 3442543.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4378.0, "completions/max_terminated_length": 4378.0, "completions/mean_length": 3682.5, "completions/mean_terminated_length": 3682.5, "completions/min_length": 2987.0, "completions/min_terminated_length": 2987.0, "epoch": 0.0155773285377651, "grad_norm": 0.0, "learning_rate": 6.865e-07, "loss": 0.0, "num_tokens": 3450720.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1259.0, "completions/max_terminated_length": 1259.0, "completions/mean_length": 1143.5, "completions/mean_terminated_length": 1143.5, "completions/min_length": 1028.0, "completions/min_terminated_length": 1028.0, "epoch": 0.015602133201041795, "grad_norm": 0.0, "learning_rate": 6.86e-07, "loss": 0.0, "num_tokens": 3454037.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 629 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015626937864318492, "grad_norm": 0.0, "learning_rate": 6.854999999999999e-07, "loss": 0.0, "num_tokens": 3454945.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 630 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 631.0, "completions/max_terminated_length": 631.0, "completions/mean_length": 491.5, "completions/mean_terminated_length": 491.5, "completions/min_length": 352.0, "completions/min_terminated_length": 352.0, "epoch": 0.01565174252759519, "grad_norm": 0.0, "learning_rate": 6.85e-07, "loss": 0.0, "num_tokens": 3456760.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5934.0, "completions/max_terminated_length": 5934.0, "completions/mean_length": 5890.5, "completions/mean_terminated_length": 5890.5, "completions/min_length": 5847.0, "completions/min_terminated_length": 5847.0, "epoch": 0.015676547190871883, "grad_norm": 0.0, "learning_rate": 6.845e-07, "loss": 0.0, "num_tokens": 3469493.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 632 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5759.0, "completions/mean_length": 6975.5, "completions/mean_terminated_length": 5759.0, "completions/min_length": 5759.0, "completions/min_terminated_length": 5759.0, "epoch": 0.01570135185414858, "grad_norm": 0.0, "learning_rate": 6.84e-07, "loss": 0.0, "num_tokens": 3476184.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4295.0, "completions/max_terminated_length": 4295.0, "completions/mean_length": 3416.5, "completions/mean_terminated_length": 3416.5, "completions/min_length": 2538.0, "completions/min_terminated_length": 2538.0, "epoch": 0.015726156517425274, "grad_norm": 0.0, "learning_rate": 6.835e-07, "loss": 0.0, "num_tokens": 3484031.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5802.0, "completions/max_terminated_length": 5802.0, "completions/mean_length": 5755.5, "completions/mean_terminated_length": 5755.5, "completions/min_length": 5709.0, "completions/min_terminated_length": 5709.0, "epoch": 0.01575096118070197, "grad_norm": 0.0, "learning_rate": 6.830000000000001e-07, "loss": 0.0, "num_tokens": 3496402.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1297.0, "completions/max_terminated_length": 1297.0, "completions/mean_length": 1106.5, "completions/mean_terminated_length": 1106.5, "completions/min_length": 916.0, "completions/min_terminated_length": 916.0, "epoch": 0.01577576584397867, "grad_norm": 0.0, "learning_rate": 6.824999999999999e-07, "loss": 0.0, "num_tokens": 3499403.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5871.0, "completions/max_terminated_length": 5871.0, "completions/mean_length": 4723.5, "completions/mean_terminated_length": 4723.5, "completions/min_length": 3576.0, "completions/min_terminated_length": 3576.0, "epoch": 0.015800570507255363, "grad_norm": 0.0, "learning_rate": 6.82e-07, "loss": 0.0, "num_tokens": 3509924.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 637 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01582537517053206, "grad_norm": 0.0, "learning_rate": 6.815e-07, "loss": 0.0, "num_tokens": 3510926.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 638 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015850179833808757, "grad_norm": 0.0, "learning_rate": 6.81e-07, "loss": 0.0, "num_tokens": 3511804.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 639 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3086.0, "completions/max_terminated_length": 3086.0, "completions/mean_length": 2946.0, "completions/mean_terminated_length": 2946.0, "completions/min_length": 2806.0, "completions/min_terminated_length": 2806.0, "epoch": 0.01587498449708545, "grad_norm": 3.7220699787139893, "learning_rate": 6.805e-07, "loss": -0.0336, "num_tokens": 3518636.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 832.0, "completions/max_terminated_length": 832.0, "completions/mean_length": 643.5, "completions/mean_terminated_length": 643.5, "completions/min_length": 455.0, "completions/min_terminated_length": 455.0, "epoch": 0.015899789160362148, "grad_norm": 0.0, "learning_rate": 6.800000000000001e-07, "loss": 0.0, "num_tokens": 3520741.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 641 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.015924593823638845, "grad_norm": 0.0, "learning_rate": 6.794999999999999e-07, "loss": 0.0, "num_tokens": 3521587.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 642 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01594939848691554, "grad_norm": 0.0, "learning_rate": 6.79e-07, "loss": 0.0, "num_tokens": 3522589.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3591.0, "completions/max_terminated_length": 3591.0, "completions/mean_length": 3387.0, "completions/mean_terminated_length": 3387.0, "completions/min_length": 3183.0, "completions/min_terminated_length": 3183.0, "epoch": 0.015974203150192236, "grad_norm": 0.0, "learning_rate": 6.784999999999999e-07, "loss": 0.0, "num_tokens": 3530227.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 721.0, "completions/max_terminated_length": 721.0, "completions/mean_length": 670.5, "completions/mean_terminated_length": 670.5, "completions/min_length": 620.0, "completions/min_terminated_length": 620.0, "epoch": 0.015999007813468934, "grad_norm": 0.0, "learning_rate": 6.78e-07, "loss": 0.0, "num_tokens": 3532396.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 645 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016023812476745627, "grad_norm": 0.0, "learning_rate": 6.775e-07, "loss": 0.0, "num_tokens": 3533318.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4850.0, "completions/mean_length": 6521.0, "completions/mean_terminated_length": 4850.0, "completions/min_length": 4850.0, "completions/min_terminated_length": 4850.0, "epoch": 0.016048617140022325, "grad_norm": 5.155060291290283, "learning_rate": 6.77e-07, "loss": -0.707, "num_tokens": 3539100.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4943.0, "completions/mean_length": 6567.5, "completions/mean_terminated_length": 4943.0, "completions/min_length": 4943.0, "completions/min_terminated_length": 4943.0, "epoch": 0.01607342180329902, "grad_norm": 2.867572069168091, "learning_rate": 6.765e-07, "loss": -0.707, "num_tokens": 3544951.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 648 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1467.0, "completions/max_terminated_length": 1467.0, "completions/mean_length": 1293.0, "completions/mean_terminated_length": 1293.0, "completions/min_length": 1119.0, "completions/min_terminated_length": 1119.0, "epoch": 0.016098226466575716, "grad_norm": 0.0, "learning_rate": 6.76e-07, "loss": 0.0, "num_tokens": 3548343.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3685.0, "completions/max_terminated_length": 3685.0, "completions/mean_length": 3272.0, "completions/mean_terminated_length": 3272.0, "completions/min_length": 2859.0, "completions/min_terminated_length": 2859.0, "epoch": 0.016123031129852413, "grad_norm": 0.0, "learning_rate": 6.754999999999999e-07, "loss": 0.0, "num_tokens": 3555729.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 650 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4230.0, "completions/max_terminated_length": 4230.0, "completions/mean_length": 3437.0, "completions/mean_terminated_length": 3437.0, "completions/min_length": 2644.0, "completions/min_terminated_length": 2644.0, "epoch": 0.016147835793129107, "grad_norm": 3.3112456798553467, "learning_rate": 6.75e-07, "loss": 0.1631, "num_tokens": 3563517.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 651 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3482.0, "completions/max_terminated_length": 3482.0, "completions/mean_length": 2537.5, "completions/mean_terminated_length": 2537.5, "completions/min_length": 1593.0, "completions/min_terminated_length": 1593.0, "epoch": 0.016172640456405804, "grad_norm": 3.291532039642334, "learning_rate": 6.745e-07, "loss": -0.2632, "num_tokens": 3569550.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 652 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0161974451196825, "grad_norm": 0.0, "learning_rate": 6.74e-07, "loss": 0.0, "num_tokens": 3570550.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2436.0, "completions/max_terminated_length": 2436.0, "completions/mean_length": 2399.0, "completions/mean_terminated_length": 2399.0, "completions/min_length": 2362.0, "completions/min_terminated_length": 2362.0, "epoch": 0.016222249782959195, "grad_norm": 0.0, "learning_rate": 6.735e-07, "loss": 0.0, "num_tokens": 3576194.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5970.0, "completions/max_terminated_length": 5970.0, "completions/mean_length": 4334.0, "completions/mean_terminated_length": 4334.0, "completions/min_length": 2698.0, "completions/min_terminated_length": 2698.0, "epoch": 0.016247054446235892, "grad_norm": 0.0, "learning_rate": 6.730000000000001e-07, "loss": 0.0, "num_tokens": 3585724.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4938.0, "completions/mean_length": 6565.0, "completions/mean_terminated_length": 4938.0, "completions/min_length": 4938.0, "completions/min_terminated_length": 4938.0, "epoch": 0.01627185910951259, "grad_norm": 0.0, "learning_rate": 6.724999999999999e-07, "loss": 0.0, "num_tokens": 3591554.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6708.0, "completions/max_terminated_length": 6708.0, "completions/mean_length": 4976.0, "completions/mean_terminated_length": 4976.0, "completions/min_length": 3244.0, "completions/min_terminated_length": 3244.0, "epoch": 0.016296663772789283, "grad_norm": 0.0, "learning_rate": 6.72e-07, "loss": 0.0, "num_tokens": 3602344.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 657 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6628.0, "completions/max_terminated_length": 6628.0, "completions/mean_length": 5394.0, "completions/mean_terminated_length": 5394.0, "completions/min_length": 4160.0, "completions/min_terminated_length": 4160.0, "epoch": 0.01632146843606598, "grad_norm": 0.0, "learning_rate": 6.714999999999999e-07, "loss": 0.0, "num_tokens": 3613972.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 992.0, "completions/max_terminated_length": 992.0, "completions/mean_length": 872.5, "completions/mean_terminated_length": 872.5, "completions/min_length": 753.0, "completions/min_terminated_length": 753.0, "epoch": 0.016346273099342678, "grad_norm": 0.0, "learning_rate": 6.71e-07, "loss": 0.0, "num_tokens": 3616533.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 659 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01637107776261937, "grad_norm": 0.0, "learning_rate": 6.705e-07, "loss": 0.0, "num_tokens": 3617541.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 660 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01639588242589607, "grad_norm": 0.0, "learning_rate": 6.7e-07, "loss": 0.0, "num_tokens": 3618407.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8170.0, "completions/max_terminated_length": 8170.0, "completions/mean_length": 7900.5, "completions/mean_terminated_length": 7900.5, "completions/min_length": 7631.0, "completions/min_terminated_length": 7631.0, "epoch": 0.016420687089172766, "grad_norm": 0.0, "learning_rate": 6.695e-07, "loss": 0.0, "num_tokens": 3635158.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5853.0, "completions/max_terminated_length": 5853.0, "completions/mean_length": 4710.0, "completions/mean_terminated_length": 4710.0, "completions/min_length": 3567.0, "completions/min_terminated_length": 3567.0, "epoch": 0.01644549175244946, "grad_norm": 2.7573533058166504, "learning_rate": 6.69e-07, "loss": -0.1716, "num_tokens": 3645532.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5930.0, "completions/max_terminated_length": 5930.0, "completions/mean_length": 5903.0, "completions/mean_terminated_length": 5903.0, "completions/min_length": 5876.0, "completions/min_terminated_length": 5876.0, "epoch": 0.016470296415726157, "grad_norm": 0.0, "learning_rate": 6.684999999999999e-07, "loss": 0.0, "num_tokens": 3658160.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4759.0, "completions/mean_length": 6475.5, "completions/mean_terminated_length": 4759.0, "completions/min_length": 4759.0, "completions/min_terminated_length": 4759.0, "epoch": 0.01649510107900285, "grad_norm": 3.239015579223633, "learning_rate": 6.68e-07, "loss": -0.707, "num_tokens": 3663783.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 665 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016519905742279548, "grad_norm": 0.0, "learning_rate": 6.675e-07, "loss": 0.0, "num_tokens": 3664683.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5498.0, "completions/max_terminated_length": 5498.0, "completions/mean_length": 4558.0, "completions/mean_terminated_length": 4558.0, "completions/min_length": 3618.0, "completions/min_terminated_length": 3618.0, "epoch": 0.016544710405556246, "grad_norm": 2.8989460468292236, "learning_rate": 6.67e-07, "loss": -0.1458, "num_tokens": 3674713.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 667 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01656951506883294, "grad_norm": 0.0, "learning_rate": 6.665e-07, "loss": 0.0, "num_tokens": 3675683.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 668 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6528.0, "completions/mean_length": 7360.0, "completions/mean_terminated_length": 6528.0, "completions/min_length": 6528.0, "completions/min_terminated_length": 6528.0, "epoch": 0.016594319732109637, "grad_norm": 0.0, "learning_rate": 6.66e-07, "loss": 0.0, "num_tokens": 3683167.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 669 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016619124395386334, "grad_norm": 0.0, "learning_rate": 6.654999999999999e-07, "loss": 0.0, "num_tokens": 3684123.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 670 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016643929058663028, "grad_norm": 0.0, "learning_rate": 6.65e-07, "loss": 0.0, "num_tokens": 3685081.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 671 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4912.0, "completions/mean_length": 6552.0, "completions/mean_terminated_length": 4912.0, "completions/min_length": 4912.0, "completions/min_terminated_length": 4912.0, "epoch": 0.016668733721939725, "grad_norm": 0.0, "learning_rate": 6.645e-07, "loss": 0.0, "num_tokens": 3690853.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8023.0, "completions/mean_length": 8107.5, "completions/mean_terminated_length": 8023.0, "completions/min_length": 8023.0, "completions/min_terminated_length": 8023.0, "epoch": 0.016693538385216422, "grad_norm": 0.0, "learning_rate": 6.64e-07, "loss": 0.0, "num_tokens": 3699846.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 673 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2088.0, "completions/max_terminated_length": 2088.0, "completions/mean_length": 1764.0, "completions/mean_terminated_length": 1764.0, "completions/min_length": 1440.0, "completions/min_terminated_length": 1440.0, "epoch": 0.016718343048493116, "grad_norm": 0.0, "learning_rate": 6.635e-07, "loss": 0.0, "num_tokens": 3704240.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2311.0, "completions/max_terminated_length": 2311.0, "completions/mean_length": 1847.5, "completions/mean_terminated_length": 1847.5, "completions/min_length": 1384.0, "completions/min_terminated_length": 1384.0, "epoch": 0.016743147711769813, "grad_norm": 0.0, "learning_rate": 6.63e-07, "loss": 0.0, "num_tokens": 3708787.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 675 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01676795237504651, "grad_norm": 0.0, "learning_rate": 6.624999999999999e-07, "loss": 0.0, "num_tokens": 3709743.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5933.0, "completions/mean_length": 7062.5, "completions/mean_terminated_length": 5933.0, "completions/min_length": 5933.0, "completions/min_terminated_length": 5933.0, "epoch": 0.016792757038323204, "grad_norm": 3.484506845474243, "learning_rate": 6.62e-07, "loss": -0.707, "num_tokens": 3716564.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 677 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5139.0, "completions/max_terminated_length": 5139.0, "completions/mean_length": 3687.5, "completions/mean_terminated_length": 3687.5, "completions/min_length": 2236.0, "completions/min_terminated_length": 2236.0, "epoch": 0.0168175617015999, "grad_norm": 0.0, "learning_rate": 6.614999999999999e-07, "loss": 0.0, "num_tokens": 3724949.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 678 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016842366364876595, "grad_norm": 0.0, "learning_rate": 6.61e-07, "loss": 0.0, "num_tokens": 3725927.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 679 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016867171028153292, "grad_norm": 0.0, "learning_rate": 6.605e-07, "loss": 0.0, "num_tokens": 3726843.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 680 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 835.0, "completions/max_terminated_length": 835.0, "completions/mean_length": 626.0, "completions/mean_terminated_length": 626.0, "completions/min_length": 417.0, "completions/min_terminated_length": 417.0, "epoch": 0.01689197569142999, "grad_norm": 0.0, "learning_rate": 6.6e-07, "loss": 0.0, "num_tokens": 3728883.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 681 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016916780354706683, "grad_norm": 0.0, "learning_rate": 6.595e-07, "loss": 0.0, "num_tokens": 3730483.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7843.0, "completions/max_terminated_length": 7843.0, "completions/mean_length": 6600.0, "completions/mean_terminated_length": 6600.0, "completions/min_length": 5357.0, "completions/min_terminated_length": 5357.0, "epoch": 0.01694158501798338, "grad_norm": 0.0, "learning_rate": 6.59e-07, "loss": 0.0, "num_tokens": 3744667.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 683 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.016966389681260078, "grad_norm": 0.0, "learning_rate": 6.584999999999999e-07, "loss": 0.0, "num_tokens": 3745573.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6051.0, "completions/max_terminated_length": 6051.0, "completions/mean_length": 4641.5, "completions/mean_terminated_length": 4641.5, "completions/min_length": 3232.0, "completions/min_terminated_length": 3232.0, "epoch": 0.016991194344536772, "grad_norm": 0.0, "learning_rate": 6.58e-07, "loss": 0.0, "num_tokens": 3755766.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 685 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01701599900781347, "grad_norm": 0.0, "learning_rate": 6.575e-07, "loss": 0.0, "num_tokens": 3756678.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 686 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4003.0, "completions/mean_length": 6097.5, "completions/mean_terminated_length": 4003.0, "completions/min_length": 4003.0, "completions/min_terminated_length": 4003.0, "epoch": 0.017040803671090166, "grad_norm": 0.0, "learning_rate": 6.57e-07, "loss": 0.0, "num_tokens": 3761537.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 687 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1366.0, "completions/max_terminated_length": 1366.0, "completions/mean_length": 1349.5, "completions/mean_terminated_length": 1349.5, "completions/min_length": 1333.0, "completions/min_terminated_length": 1333.0, "epoch": 0.01706560833436686, "grad_norm": 0.0, "learning_rate": 6.565e-07, "loss": 0.0, "num_tokens": 3765096.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 688 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1807.0, "completions/max_terminated_length": 1807.0, "completions/mean_length": 1560.5, "completions/mean_terminated_length": 1560.5, "completions/min_length": 1314.0, "completions/min_terminated_length": 1314.0, "epoch": 0.017090412997643557, "grad_norm": 0.0, "learning_rate": 6.56e-07, "loss": 0.0, "num_tokens": 3769135.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 689 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017115217660920255, "grad_norm": 0.0, "learning_rate": 6.554999999999999e-07, "loss": 0.0, "num_tokens": 3770033.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 690 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01714002232419695, "grad_norm": 0.0, "learning_rate": 6.55e-07, "loss": 0.0, "num_tokens": 3771179.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1597.0, "completions/max_terminated_length": 1597.0, "completions/mean_length": 1211.5, "completions/mean_terminated_length": 1211.5, "completions/min_length": 826.0, "completions/min_terminated_length": 826.0, "epoch": 0.017164826987473646, "grad_norm": 0.0, "learning_rate": 6.544999999999999e-07, "loss": 0.0, "num_tokens": 3774420.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5661.0, "completions/mean_length": 6926.5, "completions/mean_terminated_length": 5661.0, "completions/min_length": 5661.0, "completions/min_terminated_length": 5661.0, "epoch": 0.01718963165075034, "grad_norm": 0.0, "learning_rate": 6.54e-07, "loss": 0.0, "num_tokens": 3780961.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 693 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8049.0, "completions/max_terminated_length": 8049.0, "completions/mean_length": 6850.0, "completions/mean_terminated_length": 6850.0, "completions/min_length": 5651.0, "completions/min_terminated_length": 5651.0, "epoch": 0.017214436314027037, "grad_norm": 2.9779975414276123, "learning_rate": 6.535e-07, "loss": -0.1238, "num_tokens": 3795689.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7075.0, "completions/max_terminated_length": 7075.0, "completions/mean_length": 6629.0, "completions/mean_terminated_length": 6629.0, "completions/min_length": 6183.0, "completions/min_terminated_length": 6183.0, "epoch": 0.017239240977303734, "grad_norm": 0.0, "learning_rate": 6.53e-07, "loss": 0.0, "num_tokens": 3809903.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1706.0, "completions/max_terminated_length": 1706.0, "completions/mean_length": 1493.5, "completions/mean_terminated_length": 1493.5, "completions/min_length": 1281.0, "completions/min_terminated_length": 1281.0, "epoch": 0.017264045640580428, "grad_norm": 0.0, "learning_rate": 6.524999999999999e-07, "loss": 0.0, "num_tokens": 3813772.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4741.0, "completions/max_terminated_length": 4741.0, "completions/mean_length": 3281.5, "completions/mean_terminated_length": 3281.5, "completions/min_length": 1822.0, "completions/min_terminated_length": 1822.0, "epoch": 0.017288850303857125, "grad_norm": 0.0, "learning_rate": 6.52e-07, "loss": 0.0, "num_tokens": 3821181.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5885.0, "completions/mean_length": 7038.5, "completions/mean_terminated_length": 5885.0, "completions/min_length": 5885.0, "completions/min_terminated_length": 5885.0, "epoch": 0.017313654967133822, "grad_norm": 0.0, "learning_rate": 6.514999999999999e-07, "loss": 0.0, "num_tokens": 3827996.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 698 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6945.0, "completions/max_terminated_length": 6945.0, "completions/mean_length": 6482.5, "completions/mean_terminated_length": 6482.5, "completions/min_length": 6020.0, "completions/min_terminated_length": 6020.0, "epoch": 0.017338459630410516, "grad_norm": 2.297766923904419, "learning_rate": 6.51e-07, "loss": 0.0504, "num_tokens": 3841939.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 699 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017363264293687213, "grad_norm": 0.0, "learning_rate": 6.505e-07, "loss": 0.0, "num_tokens": 3842809.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4927.0, "completions/max_terminated_length": 4927.0, "completions/mean_length": 3313.0, "completions/mean_terminated_length": 3313.0, "completions/min_length": 1699.0, "completions/min_terminated_length": 1699.0, "epoch": 0.01738806895696391, "grad_norm": 0.0, "learning_rate": 6.5e-07, "loss": 0.0, "num_tokens": 3850347.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 701 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 557.0, "completions/max_terminated_length": 557.0, "completions/mean_length": 496.0, "completions/mean_terminated_length": 496.0, "completions/min_length": 435.0, "completions/min_terminated_length": 435.0, "epoch": 0.017412873620240604, "grad_norm": 0.0, "learning_rate": 6.495e-07, "loss": 0.0, "num_tokens": 3852187.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1457.0, "completions/max_terminated_length": 1457.0, "completions/mean_length": 1375.5, "completions/mean_terminated_length": 1375.5, "completions/min_length": 1294.0, "completions/min_terminated_length": 1294.0, "epoch": 0.0174376782835173, "grad_norm": 0.0, "learning_rate": 6.49e-07, "loss": 0.0, "num_tokens": 3855766.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6473.0, "completions/mean_length": 7332.5, "completions/mean_terminated_length": 6473.0, "completions/min_length": 6473.0, "completions/min_terminated_length": 6473.0, "epoch": 0.017462482946794, "grad_norm": 3.479776620864868, "learning_rate": 6.484999999999999e-07, "loss": -0.707, "num_tokens": 3863101.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 704 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017487287610070693, "grad_norm": 0.0, "learning_rate": 6.48e-07, "loss": 0.0, "num_tokens": 3864043.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8074.0, "completions/mean_length": 8133.0, "completions/mean_terminated_length": 8074.0, "completions/min_length": 8074.0, "completions/min_terminated_length": 8074.0, "epoch": 0.01751209227334739, "grad_norm": 0.0, "learning_rate": 6.474999999999999e-07, "loss": 0.0, "num_tokens": 3872951.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 706 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017536896936624087, "grad_norm": 0.0, "learning_rate": 6.47e-07, "loss": 0.0, "num_tokens": 3873911.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4523.0, "completions/max_terminated_length": 4523.0, "completions/mean_length": 2917.5, "completions/mean_terminated_length": 2917.5, "completions/min_length": 1312.0, "completions/min_terminated_length": 1312.0, "epoch": 0.01756170159990078, "grad_norm": 0.0, "learning_rate": 6.465e-07, "loss": 0.0, "num_tokens": 3880622.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 708 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017586506263177478, "grad_norm": 0.0, "learning_rate": 6.46e-07, "loss": 0.0, "num_tokens": 3881552.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 709 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2221.0, "completions/max_terminated_length": 2221.0, "completions/mean_length": 2137.0, "completions/mean_terminated_length": 2137.0, "completions/min_length": 2053.0, "completions/min_terminated_length": 2053.0, "epoch": 0.017611310926454172, "grad_norm": 0.0, "learning_rate": 6.454999999999999e-07, "loss": 0.0, "num_tokens": 3886632.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1490.0, "completions/max_terminated_length": 1490.0, "completions/mean_length": 1489.5, "completions/mean_terminated_length": 1489.5, "completions/min_length": 1489.0, "completions/min_terminated_length": 1489.0, "epoch": 0.01763611558973087, "grad_norm": 0.0, "learning_rate": 6.45e-07, "loss": 0.0, "num_tokens": 3890539.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4945.0, "completions/max_terminated_length": 4945.0, "completions/mean_length": 4614.5, "completions/mean_terminated_length": 4614.5, "completions/min_length": 4284.0, "completions/min_terminated_length": 4284.0, "epoch": 0.017660920253007566, "grad_norm": 3.084160566329956, "learning_rate": 6.444999999999999e-07, "loss": 0.0506, "num_tokens": 3900638.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4704.0, "completions/mean_length": 6448.0, "completions/mean_terminated_length": 4704.0, "completions/min_length": 4704.0, "completions/min_terminated_length": 4704.0, "epoch": 0.01768572491628426, "grad_norm": 0.0, "learning_rate": 6.44e-07, "loss": 0.0, "num_tokens": 3906338.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2380.0, "completions/max_terminated_length": 2380.0, "completions/mean_length": 1896.0, "completions/mean_terminated_length": 1896.0, "completions/min_length": 1412.0, "completions/min_terminated_length": 1412.0, "epoch": 0.017710529579560957, "grad_norm": 0.0, "learning_rate": 6.435e-07, "loss": 0.0, "num_tokens": 3910936.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5334.0, "completions/max_terminated_length": 5334.0, "completions/mean_length": 3666.5, "completions/mean_terminated_length": 3666.5, "completions/min_length": 1999.0, "completions/min_terminated_length": 1999.0, "epoch": 0.017735334242837655, "grad_norm": 0.0, "learning_rate": 6.43e-07, "loss": 0.0, "num_tokens": 3919125.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2155.0, "completions/max_terminated_length": 2155.0, "completions/mean_length": 2054.5, "completions/mean_terminated_length": 2054.5, "completions/min_length": 1954.0, "completions/min_terminated_length": 1954.0, "epoch": 0.01776013890611435, "grad_norm": 0.0, "learning_rate": 6.424999999999999e-07, "loss": 0.0, "num_tokens": 3924124.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4992.0, "completions/max_terminated_length": 4992.0, "completions/mean_length": 3373.5, "completions/mean_terminated_length": 3373.5, "completions/min_length": 1755.0, "completions/min_terminated_length": 1755.0, "epoch": 0.017784943569391046, "grad_norm": 0.0, "learning_rate": 6.42e-07, "loss": 0.0, "num_tokens": 3931871.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6760.0, "completions/max_terminated_length": 6760.0, "completions/mean_length": 5674.0, "completions/mean_terminated_length": 5674.0, "completions/min_length": 4588.0, "completions/min_terminated_length": 4588.0, "epoch": 0.017809748232667743, "grad_norm": 2.1770124435424805, "learning_rate": 6.414999999999999e-07, "loss": -0.1353, "num_tokens": 3944151.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 718 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5024.0, "completions/max_terminated_length": 5024.0, "completions/mean_length": 3741.0, "completions/mean_terminated_length": 3741.0, "completions/min_length": 2458.0, "completions/min_terminated_length": 2458.0, "epoch": 0.017834552895944437, "grad_norm": 0.0, "learning_rate": 6.41e-07, "loss": 0.0, "num_tokens": 3952515.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 719 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017859357559221134, "grad_norm": 0.0, "learning_rate": 6.404999999999999e-07, "loss": 0.0, "num_tokens": 3953441.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 720 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8033.0, "completions/max_terminated_length": 8033.0, "completions/mean_length": 7109.0, "completions/mean_terminated_length": 7109.0, "completions/min_length": 6185.0, "completions/min_terminated_length": 6185.0, "epoch": 0.01788416222249783, "grad_norm": 0.0, "learning_rate": 6.4e-07, "loss": 0.0, "num_tokens": 3968489.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 721 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.017908966885774525, "grad_norm": 0.0, "learning_rate": 6.395e-07, "loss": 0.0, "num_tokens": 3969429.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6897.0, "completions/max_terminated_length": 6897.0, "completions/mean_length": 5848.5, "completions/mean_terminated_length": 5848.5, "completions/min_length": 4800.0, "completions/min_terminated_length": 4800.0, "epoch": 0.017933771549051222, "grad_norm": 0.0, "learning_rate": 6.389999999999999e-07, "loss": 0.0, "num_tokens": 3982086.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3063.0, "completions/max_terminated_length": 3063.0, "completions/mean_length": 2705.5, "completions/mean_terminated_length": 2705.5, "completions/min_length": 2348.0, "completions/min_terminated_length": 2348.0, "epoch": 0.017958576212327916, "grad_norm": 0.0, "learning_rate": 6.384999999999999e-07, "loss": 0.0, "num_tokens": 3988515.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4474.0, "completions/max_terminated_length": 4474.0, "completions/mean_length": 4110.5, "completions/mean_terminated_length": 4110.5, "completions/min_length": 3747.0, "completions/min_terminated_length": 3747.0, "epoch": 0.017983380875604613, "grad_norm": 0.0, "learning_rate": 6.38e-07, "loss": 0.0, "num_tokens": 3997604.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1657.0, "completions/max_terminated_length": 1657.0, "completions/mean_length": 1460.0, "completions/mean_terminated_length": 1460.0, "completions/min_length": 1263.0, "completions/min_terminated_length": 1263.0, "epoch": 0.01800818553888131, "grad_norm": 0.0, "learning_rate": 6.374999999999999e-07, "loss": 0.0, "num_tokens": 4001440.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 726 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018032990202158004, "grad_norm": 0.0, "learning_rate": 6.37e-07, "loss": 0.0, "num_tokens": 4002304.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 727 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0180577948654347, "grad_norm": 0.0, "learning_rate": 6.365e-07, "loss": 0.0, "num_tokens": 4003268.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6249.0, "completions/mean_length": 7220.5, "completions/mean_terminated_length": 6249.0, "completions/min_length": 6249.0, "completions/min_terminated_length": 6249.0, "epoch": 0.0180825995287114, "grad_norm": 0.0, "learning_rate": 6.36e-07, "loss": 0.0, "num_tokens": 4010341.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3169.0, "completions/max_terminated_length": 3169.0, "completions/mean_length": 3069.0, "completions/mean_terminated_length": 3069.0, "completions/min_length": 2969.0, "completions/min_terminated_length": 2969.0, "epoch": 0.018107404191988093, "grad_norm": 0.0, "learning_rate": 6.354999999999999e-07, "loss": 0.0, "num_tokens": 4017421.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4733.0, "completions/mean_length": 6462.5, "completions/mean_terminated_length": 4733.0, "completions/min_length": 4733.0, "completions/min_terminated_length": 4733.0, "epoch": 0.01813220885526479, "grad_norm": 0.0, "learning_rate": 6.35e-07, "loss": 0.0, "num_tokens": 4023124.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4561.0, "completions/max_terminated_length": 4561.0, "completions/mean_length": 4029.0, "completions/mean_terminated_length": 4029.0, "completions/min_length": 3497.0, "completions/min_terminated_length": 3497.0, "epoch": 0.018157013518541487, "grad_norm": 0.0, "learning_rate": 6.344999999999999e-07, "loss": 0.0, "num_tokens": 4031998.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 732 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7084.0, "completions/max_terminated_length": 7084.0, "completions/mean_length": 5699.5, "completions/mean_terminated_length": 5699.5, "completions/min_length": 4315.0, "completions/min_terminated_length": 4315.0, "epoch": 0.01818181818181818, "grad_norm": 2.8483986854553223, "learning_rate": 6.34e-07, "loss": 0.1717, "num_tokens": 4044409.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 733 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018206622845094878, "grad_norm": 0.0, "learning_rate": 6.335e-07, "loss": 0.0, "num_tokens": 4045271.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7430.0, "completions/mean_length": 7811.0, "completions/mean_terminated_length": 7430.0, "completions/min_length": 7430.0, "completions/min_terminated_length": 7430.0, "epoch": 0.018231427508371575, "grad_norm": 0.0, "learning_rate": 6.33e-07, "loss": 0.0, "num_tokens": 4053657.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6605.0, "completions/max_terminated_length": 6605.0, "completions/mean_length": 4931.5, "completions/mean_terminated_length": 4931.5, "completions/min_length": 3258.0, "completions/min_terminated_length": 3258.0, "epoch": 0.01825623217164827, "grad_norm": 0.0, "learning_rate": 6.324999999999999e-07, "loss": 0.0, "num_tokens": 4064396.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4921.0, "completions/max_terminated_length": 4921.0, "completions/mean_length": 3486.5, "completions/mean_terminated_length": 3486.5, "completions/min_length": 2052.0, "completions/min_terminated_length": 2052.0, "epoch": 0.018281036834924966, "grad_norm": 0.0, "learning_rate": 6.319999999999999e-07, "loss": 0.0, "num_tokens": 4072335.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2677.0, "completions/max_terminated_length": 2677.0, "completions/mean_length": 2483.5, "completions/mean_terminated_length": 2483.5, "completions/min_length": 2290.0, "completions/min_terminated_length": 2290.0, "epoch": 0.01830584149820166, "grad_norm": 0.0, "learning_rate": 6.314999999999999e-07, "loss": 0.0, "num_tokens": 4078178.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7450.0, "completions/max_terminated_length": 7450.0, "completions/mean_length": 4759.0, "completions/mean_terminated_length": 4759.0, "completions/min_length": 2068.0, "completions/min_terminated_length": 2068.0, "epoch": 0.018330646161478358, "grad_norm": 0.0, "learning_rate": 6.31e-07, "loss": 0.0, "num_tokens": 4088526.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6077.0, "completions/max_terminated_length": 6077.0, "completions/mean_length": 5976.5, "completions/mean_terminated_length": 5976.5, "completions/min_length": 5876.0, "completions/min_terminated_length": 5876.0, "epoch": 0.018355450824755055, "grad_norm": 0.0, "learning_rate": 6.304999999999999e-07, "loss": 0.0, "num_tokens": 4101329.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 740 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01838025548803175, "grad_norm": 0.0, "learning_rate": 6.3e-07, "loss": 0.0, "num_tokens": 4102239.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 741 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1973.0, "completions/max_terminated_length": 1973.0, "completions/mean_length": 1609.0, "completions/mean_terminated_length": 1609.0, "completions/min_length": 1245.0, "completions/min_terminated_length": 1245.0, "epoch": 0.018405060151308446, "grad_norm": 0.0, "learning_rate": 6.295e-07, "loss": 0.0, "num_tokens": 4106279.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 742 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3054.0, "completions/max_terminated_length": 3054.0, "completions/mean_length": 2510.5, "completions/mean_terminated_length": 2510.5, "completions/min_length": 1967.0, "completions/min_terminated_length": 1967.0, "epoch": 0.018429864814585143, "grad_norm": 0.0, "learning_rate": 6.289999999999999e-07, "loss": 0.0, "num_tokens": 4112146.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 743 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018454669477861837, "grad_norm": 0.0, "learning_rate": 6.284999999999999e-07, "loss": 0.0, "num_tokens": 4113080.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 744 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018479474141138534, "grad_norm": 0.0, "learning_rate": 6.28e-07, "loss": 0.0, "num_tokens": 4114074.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1010.0, "completions/max_terminated_length": 1010.0, "completions/mean_length": 928.5, "completions/mean_terminated_length": 928.5, "completions/min_length": 847.0, "completions/min_terminated_length": 847.0, "epoch": 0.01850427880441523, "grad_norm": 0.0, "learning_rate": 6.274999999999999e-07, "loss": 0.0, "num_tokens": 4116801.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 746 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018529083467691925, "grad_norm": 0.0, "learning_rate": 6.27e-07, "loss": 0.0, "num_tokens": 4117755.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 747 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5740.0, "completions/max_terminated_length": 5740.0, "completions/mean_length": 4497.0, "completions/mean_terminated_length": 4497.0, "completions/min_length": 3254.0, "completions/min_terminated_length": 3254.0, "epoch": 0.018553888130968622, "grad_norm": 0.0, "learning_rate": 6.265e-07, "loss": 0.0, "num_tokens": 4127629.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 748 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7929.0, "completions/max_terminated_length": 7929.0, "completions/mean_length": 7477.0, "completions/mean_terminated_length": 7477.0, "completions/min_length": 7025.0, "completions/min_terminated_length": 7025.0, "epoch": 0.01857869279424532, "grad_norm": 0.0, "learning_rate": 6.26e-07, "loss": 0.0, "num_tokens": 4143427.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 749 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018603497457522013, "grad_norm": 0.0, "learning_rate": 6.254999999999999e-07, "loss": 0.0, "num_tokens": 4144405.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 750 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01862830212079871, "grad_norm": 0.0, "learning_rate": 6.249999999999999e-07, "loss": 0.0, "num_tokens": 4145293.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 751 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5646.0, "completions/max_terminated_length": 5646.0, "completions/mean_length": 5019.0, "completions/mean_terminated_length": 5019.0, "completions/min_length": 4392.0, "completions/min_terminated_length": 4392.0, "epoch": 0.018653106784075404, "grad_norm": 0.0, "learning_rate": 6.245e-07, "loss": 0.0, "num_tokens": 4156255.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6403.0, "completions/mean_length": 7297.5, "completions/mean_terminated_length": 6403.0, "completions/min_length": 6403.0, "completions/min_terminated_length": 6403.0, "epoch": 0.0186779114473521, "grad_norm": 0.0, "learning_rate": 6.24e-07, "loss": 0.0, "num_tokens": 4163676.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 753 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7959.0, "completions/mean_length": 8075.5, "completions/mean_terminated_length": 7959.0, "completions/min_length": 7959.0, "completions/min_terminated_length": 7959.0, "epoch": 0.0187027161106288, "grad_norm": 0.0, "learning_rate": 6.235e-07, "loss": 0.0, "num_tokens": 4172537.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5758.0, "completions/max_terminated_length": 5758.0, "completions/mean_length": 5153.0, "completions/mean_terminated_length": 5153.0, "completions/min_length": 4548.0, "completions/min_terminated_length": 4548.0, "epoch": 0.018727520773905493, "grad_norm": 0.0, "learning_rate": 6.23e-07, "loss": 0.0, "num_tokens": 4183705.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 755 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01875232543718219, "grad_norm": 0.0, "learning_rate": 6.225000000000001e-07, "loss": 0.0, "num_tokens": 4184835.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5484.0, "completions/mean_length": 6838.0, "completions/mean_terminated_length": 5484.0, "completions/min_length": 5484.0, "completions/min_terminated_length": 5484.0, "epoch": 0.018777130100458887, "grad_norm": 3.9258792400360107, "learning_rate": 6.219999999999999e-07, "loss": -0.707, "num_tokens": 4191183.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3123.0, "completions/max_terminated_length": 3123.0, "completions/mean_length": 3118.5, "completions/mean_terminated_length": 3118.5, "completions/min_length": 3114.0, "completions/min_terminated_length": 3114.0, "epoch": 0.01880193476373558, "grad_norm": 0.0, "learning_rate": 6.215e-07, "loss": 0.0, "num_tokens": 4198264.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1002.0, "completions/max_terminated_length": 1002.0, "completions/mean_length": 900.5, "completions/mean_terminated_length": 900.5, "completions/min_length": 799.0, "completions/min_terminated_length": 799.0, "epoch": 0.01882673942701228, "grad_norm": 0.0, "learning_rate": 6.21e-07, "loss": 0.0, "num_tokens": 4200875.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 759 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018851544090288976, "grad_norm": 0.0, "learning_rate": 6.205e-07, "loss": 0.0, "num_tokens": 4201845.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 760 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3696.0, "completions/max_terminated_length": 3696.0, "completions/mean_length": 3017.0, "completions/mean_terminated_length": 3017.0, "completions/min_length": 2338.0, "completions/min_terminated_length": 2338.0, "epoch": 0.01887634875356567, "grad_norm": 0.0, "learning_rate": 6.2e-07, "loss": 0.0, "num_tokens": 4208915.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 761 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.018901153416842367, "grad_norm": 0.0, "learning_rate": 6.195000000000001e-07, "loss": 0.0, "num_tokens": 4209781.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 762 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2110.0, "completions/max_terminated_length": 2110.0, "completions/mean_length": 1452.0, "completions/mean_terminated_length": 1452.0, "completions/min_length": 794.0, "completions/min_terminated_length": 794.0, "epoch": 0.018925958080119064, "grad_norm": 0.0, "learning_rate": 6.189999999999999e-07, "loss": 0.0, "num_tokens": 4213599.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3840.0, "completions/max_terminated_length": 3840.0, "completions/mean_length": 3325.5, "completions/mean_terminated_length": 3325.5, "completions/min_length": 2811.0, "completions/min_terminated_length": 2811.0, "epoch": 0.018950762743395758, "grad_norm": 0.0, "learning_rate": 6.185e-07, "loss": 0.0, "num_tokens": 4221228.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1109.0, "completions/max_terminated_length": 1109.0, "completions/mean_length": 1030.0, "completions/mean_terminated_length": 1030.0, "completions/min_length": 951.0, "completions/min_terminated_length": 951.0, "epoch": 0.018975567406672455, "grad_norm": 5.994223594665527, "learning_rate": 6.18e-07, "loss": -0.0542, "num_tokens": 4224148.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8169.0, "completions/max_terminated_length": 8169.0, "completions/mean_length": 7947.0, "completions/mean_terminated_length": 7947.0, "completions/min_length": 7725.0, "completions/min_terminated_length": 7725.0, "epoch": 0.019000372069949152, "grad_norm": 0.0, "learning_rate": 6.175e-07, "loss": 0.0, "num_tokens": 4240896.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 766 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019025176733225846, "grad_norm": 0.0, "learning_rate": 6.17e-07, "loss": 0.0, "num_tokens": 4241756.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 767 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019049981396502543, "grad_norm": 0.0, "learning_rate": 6.165e-07, "loss": 0.0, "num_tokens": 4242594.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 768 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4159.0, "completions/mean_length": 6175.5, "completions/mean_terminated_length": 4159.0, "completions/min_length": 4159.0, "completions/min_terminated_length": 4159.0, "epoch": 0.019074786059779237, "grad_norm": 4.115339756011963, "learning_rate": 6.16e-07, "loss": -0.707, "num_tokens": 4247701.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 769 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019099590723055934, "grad_norm": 0.0, "learning_rate": 6.155e-07, "loss": 0.0, "num_tokens": 4248617.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 770 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01912439538633263, "grad_norm": 0.0, "learning_rate": 6.149999999999999e-07, "loss": 0.0, "num_tokens": 4249693.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7203.0, "completions/mean_length": 7697.5, "completions/mean_terminated_length": 7203.0, "completions/min_length": 7203.0, "completions/min_terminated_length": 7203.0, "epoch": 0.019149200049609325, "grad_norm": 2.8248953819274902, "learning_rate": 6.145e-07, "loss": -0.707, "num_tokens": 4257784.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 772 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7840.0, "completions/mean_length": 8016.0, "completions/mean_terminated_length": 7840.0, "completions/min_length": 7840.0, "completions/min_terminated_length": 7840.0, "epoch": 0.019174004712886022, "grad_norm": 2.8419861793518066, "learning_rate": 6.14e-07, "loss": -0.707, "num_tokens": 4266704.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 773 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1435.0, "completions/max_terminated_length": 1435.0, "completions/mean_length": 1266.0, "completions/mean_terminated_length": 1266.0, "completions/min_length": 1097.0, "completions/min_terminated_length": 1097.0, "epoch": 0.01919880937616272, "grad_norm": 0.0, "learning_rate": 6.135e-07, "loss": 0.0, "num_tokens": 4270054.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6977.0, "completions/max_terminated_length": 6977.0, "completions/mean_length": 6483.5, "completions/mean_terminated_length": 6483.5, "completions/min_length": 5990.0, "completions/min_terminated_length": 5990.0, "epoch": 0.019223614039439414, "grad_norm": 0.0, "learning_rate": 6.13e-07, "loss": 0.0, "num_tokens": 4284241.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 832.0, "completions/max_terminated_length": 832.0, "completions/mean_length": 658.5, "completions/mean_terminated_length": 658.5, "completions/min_length": 485.0, "completions/min_terminated_length": 485.0, "epoch": 0.01924841870271611, "grad_norm": 0.0, "learning_rate": 6.125000000000001e-07, "loss": 0.0, "num_tokens": 4286384.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 776 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019273223365992808, "grad_norm": 0.0, "learning_rate": 6.119999999999999e-07, "loss": 0.0, "num_tokens": 4287336.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5687.0, "completions/max_terminated_length": 5687.0, "completions/mean_length": 5076.5, "completions/mean_terminated_length": 5076.5, "completions/min_length": 4466.0, "completions/min_terminated_length": 4466.0, "epoch": 0.019298028029269502, "grad_norm": 0.0, "learning_rate": 6.115e-07, "loss": 0.0, "num_tokens": 4298481.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 778 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4374.0, "completions/max_terminated_length": 4374.0, "completions/mean_length": 3381.5, "completions/mean_terminated_length": 3381.5, "completions/min_length": 2389.0, "completions/min_terminated_length": 2389.0, "epoch": 0.0193228326925462, "grad_norm": 3.1946966648101807, "learning_rate": 6.11e-07, "loss": 0.2075, "num_tokens": 4306208.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6407.0, "completions/max_terminated_length": 6407.0, "completions/mean_length": 5732.5, "completions/mean_terminated_length": 5732.5, "completions/min_length": 5058.0, "completions/min_terminated_length": 5058.0, "epoch": 0.019347637355822896, "grad_norm": 0.0, "learning_rate": 6.105e-07, "loss": 0.0, "num_tokens": 4318551.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4648.0, "completions/max_terminated_length": 4648.0, "completions/mean_length": 2687.0, "completions/mean_terminated_length": 2687.0, "completions/min_length": 726.0, "completions/min_terminated_length": 726.0, "epoch": 0.01937244201909959, "grad_norm": 0.0, "learning_rate": 6.1e-07, "loss": 0.0, "num_tokens": 4324789.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 781 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2060.0, "completions/mean_length": 5126.0, "completions/mean_terminated_length": 2060.0, "completions/min_length": 2060.0, "completions/min_terminated_length": 2060.0, "epoch": 0.019397246682376287, "grad_norm": 0.0, "learning_rate": 6.095e-07, "loss": 0.0, "num_tokens": 4327857.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 782 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2149.0, "completions/max_terminated_length": 2149.0, "completions/mean_length": 1631.5, "completions/mean_terminated_length": 1631.5, "completions/min_length": 1114.0, "completions/min_terminated_length": 1114.0, "epoch": 0.01942205134565298, "grad_norm": 0.0, "learning_rate": 6.089999999999999e-07, "loss": 0.0, "num_tokens": 4331978.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8064.0, "completions/max_terminated_length": 8064.0, "completions/mean_length": 6234.5, "completions/mean_terminated_length": 6234.5, "completions/min_length": 4405.0, "completions/min_terminated_length": 4405.0, "epoch": 0.01944685600892968, "grad_norm": 0.0, "learning_rate": 6.085e-07, "loss": 0.0, "num_tokens": 4345467.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2657.0, "completions/max_terminated_length": 2657.0, "completions/mean_length": 2107.0, "completions/mean_terminated_length": 2107.0, "completions/min_length": 1557.0, "completions/min_terminated_length": 1557.0, "epoch": 0.019471660672206376, "grad_norm": 0.0, "learning_rate": 6.079999999999999e-07, "loss": 0.0, "num_tokens": 4350569.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5227.0, "completions/max_terminated_length": 5227.0, "completions/mean_length": 5059.5, "completions/mean_terminated_length": 5059.5, "completions/min_length": 4892.0, "completions/min_terminated_length": 4892.0, "epoch": 0.01949646533548307, "grad_norm": 0.0, "learning_rate": 6.075e-07, "loss": 0.0, "num_tokens": 4361610.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4072.0, "completions/max_terminated_length": 4072.0, "completions/mean_length": 3623.5, "completions/mean_terminated_length": 3623.5, "completions/min_length": 3175.0, "completions/min_terminated_length": 3175.0, "epoch": 0.019521269998759767, "grad_norm": 0.0, "learning_rate": 6.07e-07, "loss": 0.0, "num_tokens": 4369821.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2174.0, "completions/mean_length": 5183.0, "completions/mean_terminated_length": 2174.0, "completions/min_length": 2174.0, "completions/min_terminated_length": 2174.0, "epoch": 0.019546074662036464, "grad_norm": 5.0660600662231445, "learning_rate": 6.065e-07, "loss": -0.707, "num_tokens": 4372891.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2799.0, "completions/max_terminated_length": 2799.0, "completions/mean_length": 2314.5, "completions/mean_terminated_length": 2314.5, "completions/min_length": 1830.0, "completions/min_terminated_length": 1830.0, "epoch": 0.019570879325313158, "grad_norm": 0.0, "learning_rate": 6.06e-07, "loss": 0.0, "num_tokens": 4378334.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 789 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019595683988589855, "grad_norm": 0.0, "learning_rate": 6.055e-07, "loss": 0.0, "num_tokens": 4379264.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 790 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019620488651866552, "grad_norm": 0.0, "learning_rate": 6.049999999999999e-07, "loss": 0.0, "num_tokens": 4380258.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6711.0, "completions/max_terminated_length": 6711.0, "completions/mean_length": 4460.0, "completions/mean_terminated_length": 4460.0, "completions/min_length": 2209.0, "completions/min_terminated_length": 2209.0, "epoch": 0.019645293315143246, "grad_norm": 0.0, "learning_rate": 6.045e-07, "loss": 0.0, "num_tokens": 4390054.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 792 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5876.0, "completions/mean_length": 7034.0, "completions/mean_terminated_length": 5876.0, "completions/min_length": 5876.0, "completions/min_terminated_length": 5876.0, "epoch": 0.019670097978419943, "grad_norm": 0.0, "learning_rate": 6.04e-07, "loss": 0.0, "num_tokens": 4396822.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 793 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.01969490264169664, "grad_norm": 0.0, "learning_rate": 6.035e-07, "loss": 0.0, "num_tokens": 4397734.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1606.0, "completions/mean_length": 4899.0, "completions/mean_terminated_length": 1606.0, "completions/min_length": 1606.0, "completions/min_terminated_length": 1606.0, "epoch": 0.019719707304973334, "grad_norm": 5.8268842697143555, "learning_rate": 6.03e-07, "loss": -0.707, "num_tokens": 4400258.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7232.0, "completions/max_terminated_length": 7232.0, "completions/mean_length": 7126.5, "completions/mean_terminated_length": 7126.5, "completions/min_length": 7021.0, "completions/min_terminated_length": 7021.0, "epoch": 0.01974451196825003, "grad_norm": 0.0, "learning_rate": 6.025000000000001e-07, "loss": 0.0, "num_tokens": 4415535.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7103.0, "completions/max_terminated_length": 7103.0, "completions/mean_length": 4104.0, "completions/mean_terminated_length": 4104.0, "completions/min_length": 1105.0, "completions/min_terminated_length": 1105.0, "epoch": 0.019769316631526725, "grad_norm": 0.0, "learning_rate": 6.019999999999999e-07, "loss": 0.0, "num_tokens": 4424561.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 797 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1457.0, "completions/max_terminated_length": 1457.0, "completions/mean_length": 1248.5, "completions/mean_terminated_length": 1248.5, "completions/min_length": 1040.0, "completions/min_terminated_length": 1040.0, "epoch": 0.019794121294803423, "grad_norm": 0.0, "learning_rate": 6.015e-07, "loss": 0.0, "num_tokens": 4427864.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6347.0, "completions/max_terminated_length": 6347.0, "completions/mean_length": 4764.5, "completions/mean_terminated_length": 4764.5, "completions/min_length": 3182.0, "completions/min_terminated_length": 3182.0, "epoch": 0.01981892595808012, "grad_norm": 0.0, "learning_rate": 6.009999999999999e-07, "loss": 0.0, "num_tokens": 4438215.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 799 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.019843730621356814, "grad_norm": 0.0, "learning_rate": 6.005e-07, "loss": 0.0, "num_tokens": 4439073.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5211.0, "completions/mean_length": 6701.5, "completions/mean_terminated_length": 5211.0, "completions/min_length": 5211.0, "completions/min_terminated_length": 5211.0, "epoch": 0.01986853528463351, "grad_norm": 4.556071758270264, "learning_rate": 6e-07, "loss": -0.707, "num_tokens": 4445366.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5831.0, "completions/max_terminated_length": 5831.0, "completions/mean_length": 5251.5, "completions/mean_terminated_length": 5251.5, "completions/min_length": 4672.0, "completions/min_terminated_length": 4672.0, "epoch": 0.019893339947910208, "grad_norm": 0.0, "learning_rate": 5.995e-07, "loss": 0.0, "num_tokens": 4456771.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 802 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3784.0, "completions/mean_length": 5988.0, "completions/mean_terminated_length": 3784.0, "completions/min_length": 3784.0, "completions/min_terminated_length": 3784.0, "epoch": 0.019918144611186902, "grad_norm": 0.0, "learning_rate": 5.989999999999999e-07, "loss": 0.0, "num_tokens": 4461453.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 803 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7477.0, "completions/mean_length": 7834.5, "completions/mean_terminated_length": 7477.0, "completions/min_length": 7477.0, "completions/min_terminated_length": 7477.0, "epoch": 0.0199429492744636, "grad_norm": 2.5706915855407715, "learning_rate": 5.985e-07, "loss": -0.707, "num_tokens": 4470010.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3664.0, "completions/max_terminated_length": 3664.0, "completions/mean_length": 2544.0, "completions/mean_terminated_length": 2544.0, "completions/min_length": 1424.0, "completions/min_terminated_length": 1424.0, "epoch": 0.019967753937740296, "grad_norm": 0.0, "learning_rate": 5.979999999999999e-07, "loss": 0.0, "num_tokens": 4476104.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6760.0, "completions/mean_length": 7476.0, "completions/mean_terminated_length": 6760.0, "completions/min_length": 6760.0, "completions/min_terminated_length": 6760.0, "epoch": 0.01999255860101699, "grad_norm": 0.0, "learning_rate": 5.975e-07, "loss": 0.0, "num_tokens": 4483708.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 806 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.020017363264293687, "grad_norm": 0.0, "learning_rate": 5.97e-07, "loss": 0.0, "num_tokens": 4484902.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3441.0, "completions/max_terminated_length": 3441.0, "completions/mean_length": 2654.5, "completions/mean_terminated_length": 2654.5, "completions/min_length": 1868.0, "completions/min_terminated_length": 1868.0, "epoch": 0.020042167927570385, "grad_norm": 0.0, "learning_rate": 5.965e-07, "loss": 0.0, "num_tokens": 4491067.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3813.0, "completions/mean_length": 6002.5, "completions/mean_terminated_length": 3813.0, "completions/min_length": 3813.0, "completions/min_terminated_length": 3813.0, "epoch": 0.02006697259084708, "grad_norm": 4.395847320556641, "learning_rate": 5.96e-07, "loss": -0.707, "num_tokens": 4495732.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7229.0, "completions/max_terminated_length": 7229.0, "completions/mean_length": 6776.0, "completions/mean_terminated_length": 6776.0, "completions/min_length": 6323.0, "completions/min_terminated_length": 6323.0, "epoch": 0.020091777254123776, "grad_norm": 0.0, "learning_rate": 5.955e-07, "loss": 0.0, "num_tokens": 4511332.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 810 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3754.0, "completions/max_terminated_length": 3754.0, "completions/mean_length": 2262.0, "completions/mean_terminated_length": 2262.0, "completions/min_length": 770.0, "completions/min_terminated_length": 770.0, "epoch": 0.020116581917400473, "grad_norm": 0.0, "learning_rate": 5.949999999999999e-07, "loss": 0.0, "num_tokens": 4516710.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 169.0, "completions/mean_length": 4180.5, "completions/mean_terminated_length": 169.0, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "epoch": 0.020141386580677167, "grad_norm": 0.0, "learning_rate": 5.945e-07, "loss": 0.0, "num_tokens": 4517827.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2925.0, "completions/max_terminated_length": 2925.0, "completions/mean_length": 2174.0, "completions/mean_terminated_length": 2174.0, "completions/min_length": 1423.0, "completions/min_terminated_length": 1423.0, "epoch": 0.020166191243953864, "grad_norm": 0.0, "learning_rate": 5.939999999999999e-07, "loss": 0.0, "num_tokens": 4522971.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 813 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2658.0, "completions/max_terminated_length": 2658.0, "completions/mean_length": 2523.5, "completions/mean_terminated_length": 2523.5, "completions/min_length": 2389.0, "completions/min_terminated_length": 2389.0, "epoch": 0.020190995907230558, "grad_norm": 0.0, "learning_rate": 5.935e-07, "loss": 0.0, "num_tokens": 4528850.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6326.0, "completions/mean_length": 7259.0, "completions/mean_terminated_length": 6326.0, "completions/min_length": 6326.0, "completions/min_terminated_length": 6326.0, "epoch": 0.020215800570507255, "grad_norm": 3.0449278354644775, "learning_rate": 5.93e-07, "loss": -0.707, "num_tokens": 4536068.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6014.0, "completions/max_terminated_length": 6014.0, "completions/mean_length": 4368.0, "completions/mean_terminated_length": 4368.0, "completions/min_length": 2722.0, "completions/min_terminated_length": 2722.0, "epoch": 0.020240605233783952, "grad_norm": 0.0, "learning_rate": 5.925e-07, "loss": 0.0, "num_tokens": 4545678.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 816 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5608.0, "completions/max_terminated_length": 5608.0, "completions/mean_length": 5109.0, "completions/mean_terminated_length": 5109.0, "completions/min_length": 4610.0, "completions/min_terminated_length": 4610.0, "epoch": 0.020265409897060646, "grad_norm": 0.0, "learning_rate": 5.919999999999999e-07, "loss": 0.0, "num_tokens": 4556762.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6667.0, "completions/max_terminated_length": 6667.0, "completions/mean_length": 5990.0, "completions/mean_terminated_length": 5990.0, "completions/min_length": 5313.0, "completions/min_terminated_length": 5313.0, "epoch": 0.020290214560337343, "grad_norm": 2.603201150894165, "learning_rate": 5.915e-07, "loss": 0.0799, "num_tokens": 4569634.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 818 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02031501922361404, "grad_norm": 0.0, "learning_rate": 5.909999999999999e-07, "loss": 0.0, "num_tokens": 4570452.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1470.0, "completions/max_terminated_length": 1470.0, "completions/mean_length": 1175.5, "completions/mean_terminated_length": 1175.5, "completions/min_length": 881.0, "completions/min_terminated_length": 881.0, "epoch": 0.020339823886890734, "grad_norm": 0.0, "learning_rate": 5.905e-07, "loss": 0.0, "num_tokens": 4573607.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7965.0, "completions/max_terminated_length": 7965.0, "completions/mean_length": 7133.0, "completions/mean_terminated_length": 7133.0, "completions/min_length": 6301.0, "completions/min_terminated_length": 6301.0, "epoch": 0.02036462855016743, "grad_norm": 0.0, "learning_rate": 5.9e-07, "loss": 0.0, "num_tokens": 4588931.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7467.0, "completions/mean_length": 7829.5, "completions/mean_terminated_length": 7467.0, "completions/min_length": 7467.0, "completions/min_terminated_length": 7467.0, "epoch": 0.02038943321344413, "grad_norm": 0.0, "learning_rate": 5.895e-07, "loss": 0.0, "num_tokens": 4597662.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6278.0, "completions/max_terminated_length": 6278.0, "completions/mean_length": 5971.5, "completions/mean_terminated_length": 5971.5, "completions/min_length": 5665.0, "completions/min_terminated_length": 5665.0, "epoch": 0.020414237876720823, "grad_norm": 3.2465710639953613, "learning_rate": 5.89e-07, "loss": 0.0363, "num_tokens": 4610503.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 823 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02043904253999752, "grad_norm": 0.0, "learning_rate": 5.885e-07, "loss": 0.0, "num_tokens": 4611395.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4827.0, "completions/max_terminated_length": 4827.0, "completions/mean_length": 4590.5, "completions/mean_terminated_length": 4590.5, "completions/min_length": 4354.0, "completions/min_terminated_length": 4354.0, "epoch": 0.020463847203274217, "grad_norm": 0.0, "learning_rate": 5.879999999999999e-07, "loss": 0.0, "num_tokens": 4621728.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7226.0, "completions/max_terminated_length": 7226.0, "completions/mean_length": 6580.5, "completions/mean_terminated_length": 6580.5, "completions/min_length": 5935.0, "completions/min_terminated_length": 5935.0, "epoch": 0.02048865186655091, "grad_norm": 0.0, "learning_rate": 5.875e-07, "loss": 0.0, "num_tokens": 4635851.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8088.0, "completions/max_terminated_length": 8088.0, "completions/mean_length": 7770.0, "completions/mean_terminated_length": 7770.0, "completions/min_length": 7452.0, "completions/min_terminated_length": 7452.0, "epoch": 0.020513456529827608, "grad_norm": 1.8976023197174072, "learning_rate": 5.87e-07, "loss": -0.0289, "num_tokens": 4652257.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 827 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.020538261193104302, "grad_norm": 0.0, "learning_rate": 5.865e-07, "loss": 0.0, "num_tokens": 4653253.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1556.0, "completions/max_terminated_length": 1556.0, "completions/mean_length": 1470.5, "completions/mean_terminated_length": 1470.5, "completions/min_length": 1385.0, "completions/min_terminated_length": 1385.0, "epoch": 0.020563065856381, "grad_norm": 0.0, "learning_rate": 5.86e-07, "loss": 0.0, "num_tokens": 4657122.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 829 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2156.0, "completions/max_terminated_length": 2156.0, "completions/mean_length": 2042.0, "completions/mean_terminated_length": 2042.0, "completions/min_length": 1928.0, "completions/min_terminated_length": 1928.0, "epoch": 0.020587870519657697, "grad_norm": 0.0, "learning_rate": 5.854999999999999e-07, "loss": 0.0, "num_tokens": 4662140.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 830 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2826.0, "completions/max_terminated_length": 2826.0, "completions/mean_length": 2637.0, "completions/mean_terminated_length": 2637.0, "completions/min_length": 2448.0, "completions/min_terminated_length": 2448.0, "epoch": 0.02061267518293439, "grad_norm": 0.0, "learning_rate": 5.849999999999999e-07, "loss": 0.0, "num_tokens": 4668264.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 831 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.020637479846211088, "grad_norm": 0.0, "learning_rate": 5.845e-07, "loss": 0.0, "num_tokens": 4669270.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 832 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.020662284509487785, "grad_norm": 0.0, "learning_rate": 5.839999999999999e-07, "loss": 0.0, "num_tokens": 4670240.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6639.0, "completions/mean_length": 7415.5, "completions/mean_terminated_length": 6639.0, "completions/min_length": 6639.0, "completions/min_terminated_length": 6639.0, "epoch": 0.02068708917276448, "grad_norm": 3.0896356105804443, "learning_rate": 5.835e-07, "loss": -0.707, "num_tokens": 4677825.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6308.0, "completions/max_terminated_length": 6308.0, "completions/mean_length": 3649.5, "completions/mean_terminated_length": 3649.5, "completions/min_length": 991.0, "completions/min_terminated_length": 991.0, "epoch": 0.020711893836041176, "grad_norm": 0.0, "learning_rate": 5.83e-07, "loss": 0.0, "num_tokens": 4686108.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7332.0, "completions/max_terminated_length": 7332.0, "completions/mean_length": 7047.0, "completions/mean_terminated_length": 7047.0, "completions/min_length": 6762.0, "completions/min_terminated_length": 6762.0, "epoch": 0.020736698499317873, "grad_norm": 0.0, "learning_rate": 5.825e-07, "loss": 0.0, "num_tokens": 4701002.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 836 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.020761503162594567, "grad_norm": 0.0, "learning_rate": 5.819999999999999e-07, "loss": 0.0, "num_tokens": 4702098.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 837 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7380.0, "completions/max_terminated_length": 7380.0, "completions/mean_length": 5738.5, "completions/mean_terminated_length": 5738.5, "completions/min_length": 4097.0, "completions/min_terminated_length": 4097.0, "epoch": 0.020786307825871264, "grad_norm": 0.0, "learning_rate": 5.815e-07, "loss": 0.0, "num_tokens": 4714547.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1594.0, "completions/max_terminated_length": 1594.0, "completions/mean_length": 1520.0, "completions/mean_terminated_length": 1520.0, "completions/min_length": 1446.0, "completions/min_terminated_length": 1446.0, "epoch": 0.02081111248914796, "grad_norm": 0.0, "learning_rate": 5.809999999999999e-07, "loss": 0.0, "num_tokens": 4718403.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 839 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1939.0, "completions/max_terminated_length": 1939.0, "completions/mean_length": 1539.5, "completions/mean_terminated_length": 1539.5, "completions/min_length": 1140.0, "completions/min_terminated_length": 1140.0, "epoch": 0.020835917152424655, "grad_norm": 0.0, "learning_rate": 5.805e-07, "loss": 0.0, "num_tokens": 4722384.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 840 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.020860721815701352, "grad_norm": 0.0, "learning_rate": 5.8e-07, "loss": 0.0, "num_tokens": 4723382.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7248.0, "completions/mean_length": 7720.0, "completions/mean_terminated_length": 7248.0, "completions/min_length": 7248.0, "completions/min_terminated_length": 7248.0, "epoch": 0.020885526478978046, "grad_norm": 0.0, "learning_rate": 5.795e-07, "loss": 0.0, "num_tokens": 4731628.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1697.0, "completions/max_terminated_length": 1697.0, "completions/mean_length": 1500.0, "completions/mean_terminated_length": 1500.0, "completions/min_length": 1303.0, "completions/min_terminated_length": 1303.0, "epoch": 0.020910331142254743, "grad_norm": 0.0, "learning_rate": 5.79e-07, "loss": 0.0, "num_tokens": 4735432.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 843 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02093513580553144, "grad_norm": 0.0, "learning_rate": 5.784999999999999e-07, "loss": 0.0, "num_tokens": 4736716.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7210.0, "completions/max_terminated_length": 7210.0, "completions/mean_length": 6064.0, "completions/mean_terminated_length": 6064.0, "completions/min_length": 4918.0, "completions/min_terminated_length": 4918.0, "epoch": 0.020959940468808134, "grad_norm": 0.0, "learning_rate": 5.779999999999999e-07, "loss": 0.0, "num_tokens": 4749750.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1106.0, "completions/max_terminated_length": 1106.0, "completions/mean_length": 861.0, "completions/mean_terminated_length": 861.0, "completions/min_length": 616.0, "completions/min_terminated_length": 616.0, "epoch": 0.020984745132084832, "grad_norm": 0.0, "learning_rate": 5.775e-07, "loss": 0.0, "num_tokens": 4752296.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3004.0, "completions/max_terminated_length": 3004.0, "completions/mean_length": 2797.5, "completions/mean_terminated_length": 2797.5, "completions/min_length": 2591.0, "completions/min_terminated_length": 2591.0, "epoch": 0.02100954979536153, "grad_norm": 0.0, "learning_rate": 5.769999999999999e-07, "loss": 0.0, "num_tokens": 4758751.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 847 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021034354458638223, "grad_norm": 0.0, "learning_rate": 5.765e-07, "loss": 0.0, "num_tokens": 4759759.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 848 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02105915912191492, "grad_norm": 0.0, "learning_rate": 5.76e-07, "loss": 0.0, "num_tokens": 4760699.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 849 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021083963785191617, "grad_norm": 0.0, "learning_rate": 5.755e-07, "loss": 0.0, "num_tokens": 4761657.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 850 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1958.0, "completions/max_terminated_length": 1958.0, "completions/mean_length": 1738.0, "completions/mean_terminated_length": 1738.0, "completions/min_length": 1518.0, "completions/min_terminated_length": 1518.0, "epoch": 0.02110876844846831, "grad_norm": 0.0, "learning_rate": 5.749999999999999e-07, "loss": 0.0, "num_tokens": 4766297.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 851 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4455.0, "completions/mean_length": 6323.5, "completions/mean_terminated_length": 4455.0, "completions/min_length": 4455.0, "completions/min_terminated_length": 4455.0, "epoch": 0.02113357311174501, "grad_norm": 0.0, "learning_rate": 5.745e-07, "loss": 0.0, "num_tokens": 4771680.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5643.0, "completions/mean_length": 6917.5, "completions/mean_terminated_length": 5643.0, "completions/min_length": 5643.0, "completions/min_terminated_length": 5643.0, "epoch": 0.021158377775021706, "grad_norm": 0.0, "learning_rate": 5.739999999999999e-07, "loss": 0.0, "num_tokens": 4778291.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 853 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6407.0, "completions/max_terminated_length": 6407.0, "completions/mean_length": 6208.5, "completions/mean_terminated_length": 6208.5, "completions/min_length": 6010.0, "completions/min_terminated_length": 6010.0, "epoch": 0.0211831824382984, "grad_norm": 0.0, "learning_rate": 5.735e-07, "loss": 0.0, "num_tokens": 4791504.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1784.0, "completions/mean_length": 4988.0, "completions/mean_terminated_length": 1784.0, "completions/min_length": 1784.0, "completions/min_terminated_length": 1784.0, "epoch": 0.021207987101575097, "grad_norm": 6.5959062576293945, "learning_rate": 5.73e-07, "loss": -0.707, "num_tokens": 4794176.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3790.0, "completions/mean_length": 5991.0, "completions/mean_terminated_length": 3790.0, "completions/min_length": 3790.0, "completions/min_terminated_length": 3790.0, "epoch": 0.021232791764851794, "grad_norm": 3.614370346069336, "learning_rate": 5.725e-07, "loss": -0.707, "num_tokens": 4798958.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5810.0, "completions/max_terminated_length": 5810.0, "completions/mean_length": 3652.5, "completions/mean_terminated_length": 3652.5, "completions/min_length": 1495.0, "completions/min_terminated_length": 1495.0, "epoch": 0.021257596428128488, "grad_norm": 2.491562604904175, "learning_rate": 5.719999999999999e-07, "loss": -0.4176, "num_tokens": 4807289.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 857 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 785.0, "completions/max_terminated_length": 785.0, "completions/mean_length": 626.5, "completions/mean_terminated_length": 626.5, "completions/min_length": 468.0, "completions/min_terminated_length": 468.0, "epoch": 0.021282401091405185, "grad_norm": 0.0, "learning_rate": 5.715e-07, "loss": 0.0, "num_tokens": 4809354.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 886.0, "completions/max_terminated_length": 886.0, "completions/mean_length": 759.5, "completions/mean_terminated_length": 759.5, "completions/min_length": 633.0, "completions/min_terminated_length": 633.0, "epoch": 0.02130720575468188, "grad_norm": 0.0, "learning_rate": 5.709999999999999e-07, "loss": 0.0, "num_tokens": 4811737.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4952.0, "completions/max_terminated_length": 4952.0, "completions/mean_length": 4712.5, "completions/mean_terminated_length": 4712.5, "completions/min_length": 4473.0, "completions/min_terminated_length": 4473.0, "epoch": 0.021332010417958576, "grad_norm": 2.9675381183624268, "learning_rate": 5.705e-07, "loss": -0.0359, "num_tokens": 4822054.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4740.0, "completions/mean_length": 6466.0, "completions/mean_terminated_length": 4740.0, "completions/min_length": 4740.0, "completions/min_terminated_length": 4740.0, "epoch": 0.021356815081235273, "grad_norm": 0.0, "learning_rate": 5.699999999999999e-07, "loss": 0.0, "num_tokens": 4827664.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7484.0, "completions/max_terminated_length": 7484.0, "completions/mean_length": 7100.0, "completions/mean_terminated_length": 7100.0, "completions/min_length": 6716.0, "completions/min_terminated_length": 6716.0, "epoch": 0.021381619744511967, "grad_norm": 0.0, "learning_rate": 5.695e-07, "loss": 0.0, "num_tokens": 4842740.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1626.0, "completions/mean_length": 4909.0, "completions/mean_terminated_length": 1626.0, "completions/min_length": 1626.0, "completions/min_terminated_length": 1626.0, "epoch": 0.021406424407788664, "grad_norm": 0.0, "learning_rate": 5.69e-07, "loss": 0.0, "num_tokens": 4845424.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6655.0, "completions/mean_length": 7423.5, "completions/mean_terminated_length": 6655.0, "completions/min_length": 6655.0, "completions/min_terminated_length": 6655.0, "epoch": 0.02143122907106536, "grad_norm": 0.0, "learning_rate": 5.684999999999999e-07, "loss": 0.0, "num_tokens": 4852995.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 864 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021456033734342055, "grad_norm": 0.0, "learning_rate": 5.679999999999999e-07, "loss": 0.0, "num_tokens": 4854131.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2401.0, "completions/mean_length": 5296.5, "completions/mean_terminated_length": 2401.0, "completions/min_length": 2401.0, "completions/min_terminated_length": 2401.0, "epoch": 0.021480838397618753, "grad_norm": 0.0, "learning_rate": 5.675e-07, "loss": 0.0, "num_tokens": 4857362.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 866 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02150564306089545, "grad_norm": 0.0, "learning_rate": 5.669999999999999e-07, "loss": 0.0, "num_tokens": 4858272.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2392.0, "completions/max_terminated_length": 2392.0, "completions/mean_length": 2309.0, "completions/mean_terminated_length": 2309.0, "completions/min_length": 2226.0, "completions/min_terminated_length": 2226.0, "epoch": 0.021530447724172144, "grad_norm": 0.0, "learning_rate": 5.665e-07, "loss": 0.0, "num_tokens": 4863760.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2329.0, "completions/max_terminated_length": 2329.0, "completions/mean_length": 1578.5, "completions/mean_terminated_length": 1578.5, "completions/min_length": 828.0, "completions/min_terminated_length": 828.0, "epoch": 0.02155525238744884, "grad_norm": 0.0, "learning_rate": 5.66e-07, "loss": 0.0, "num_tokens": 4867817.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 869 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021580057050725538, "grad_norm": 0.0, "learning_rate": 5.655e-07, "loss": 0.0, "num_tokens": 4868703.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 870 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1010.0, "completions/max_terminated_length": 1010.0, "completions/mean_length": 791.0, "completions/mean_terminated_length": 791.0, "completions/min_length": 572.0, "completions/min_terminated_length": 572.0, "epoch": 0.021604861714002232, "grad_norm": 0.0, "learning_rate": 5.649999999999999e-07, "loss": 0.0, "num_tokens": 4871069.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 871 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 803.0, "completions/max_terminated_length": 803.0, "completions/mean_length": 796.5, "completions/mean_terminated_length": 796.5, "completions/min_length": 790.0, "completions/min_terminated_length": 790.0, "epoch": 0.02162966637727893, "grad_norm": 0.0, "learning_rate": 5.645e-07, "loss": 0.0, "num_tokens": 4873562.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 872 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8000.0, "completions/max_terminated_length": 8000.0, "completions/mean_length": 7276.5, "completions/mean_terminated_length": 7276.5, "completions/min_length": 6553.0, "completions/min_terminated_length": 6553.0, "epoch": 0.021654471040555623, "grad_norm": 2.5405149459838867, "learning_rate": 5.639999999999999e-07, "loss": 0.0703, "num_tokens": 4889007.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 873 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7916.0, "completions/mean_length": 8054.0, "completions/mean_terminated_length": 7916.0, "completions/min_length": 7916.0, "completions/min_terminated_length": 7916.0, "epoch": 0.02167927570383232, "grad_norm": 2.712531805038452, "learning_rate": 5.635e-07, "loss": -0.707, "num_tokens": 4897809.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2611.0, "completions/mean_length": 5401.5, "completions/mean_terminated_length": 2611.0, "completions/min_length": 2611.0, "completions/min_terminated_length": 2611.0, "epoch": 0.021704080367109017, "grad_norm": 4.918966770172119, "learning_rate": 5.629999999999999e-07, "loss": -0.707, "num_tokens": 4901282.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2251.0, "completions/mean_length": 5221.5, "completions/mean_terminated_length": 2251.0, "completions/min_length": 2251.0, "completions/min_terminated_length": 2251.0, "epoch": 0.02172888503038571, "grad_norm": 4.74021053314209, "learning_rate": 5.625e-07, "loss": -0.707, "num_tokens": 4904549.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 856.0, "completions/max_terminated_length": 856.0, "completions/mean_length": 822.5, "completions/mean_terminated_length": 822.5, "completions/min_length": 789.0, "completions/min_terminated_length": 789.0, "epoch": 0.02175368969366241, "grad_norm": 0.0, "learning_rate": 5.620000000000001e-07, "loss": 0.0, "num_tokens": 4907104.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 877 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021778494356939106, "grad_norm": 0.0, "learning_rate": 5.614999999999999e-07, "loss": 0.0, "num_tokens": 4908216.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 878 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4038.0, "completions/max_terminated_length": 4038.0, "completions/mean_length": 2858.0, "completions/mean_terminated_length": 2858.0, "completions/min_length": 1678.0, "completions/min_terminated_length": 1678.0, "epoch": 0.0218032990202158, "grad_norm": 3.3442437648773193, "learning_rate": 5.61e-07, "loss": -0.2919, "num_tokens": 4914756.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 879 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3441.0, "completions/max_terminated_length": 3441.0, "completions/mean_length": 2992.5, "completions/mean_terminated_length": 2992.5, "completions/min_length": 2544.0, "completions/min_terminated_length": 2544.0, "epoch": 0.021828103683492497, "grad_norm": 0.0, "learning_rate": 5.605e-07, "loss": 0.0, "num_tokens": 4921693.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 880 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4020.0, "completions/max_terminated_length": 4020.0, "completions/mean_length": 3432.5, "completions/mean_terminated_length": 3432.5, "completions/min_length": 2845.0, "completions/min_terminated_length": 2845.0, "epoch": 0.021852908346769194, "grad_norm": 2.9927241802215576, "learning_rate": 5.6e-07, "loss": -0.121, "num_tokens": 4929744.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 881 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021877713010045888, "grad_norm": 0.0, "learning_rate": 5.595e-07, "loss": 0.0, "num_tokens": 4930652.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 882 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6171.0, "completions/max_terminated_length": 6171.0, "completions/mean_length": 4707.5, "completions/mean_terminated_length": 4707.5, "completions/min_length": 3244.0, "completions/min_terminated_length": 3244.0, "epoch": 0.021902517673322585, "grad_norm": 0.0, "learning_rate": 5.590000000000001e-07, "loss": 0.0, "num_tokens": 4940949.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 883 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.021927322336599282, "grad_norm": 0.0, "learning_rate": 5.584999999999999e-07, "loss": 0.0, "num_tokens": 4941911.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1031.0, "completions/max_terminated_length": 1031.0, "completions/mean_length": 852.0, "completions/mean_terminated_length": 852.0, "completions/min_length": 673.0, "completions/min_terminated_length": 673.0, "epoch": 0.021952126999875976, "grad_norm": 0.0, "learning_rate": 5.58e-07, "loss": 0.0, "num_tokens": 4944411.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6367.0, "completions/mean_length": 7279.5, "completions/mean_terminated_length": 6367.0, "completions/min_length": 6367.0, "completions/min_terminated_length": 6367.0, "epoch": 0.021976931663152673, "grad_norm": 0.0, "learning_rate": 5.575e-07, "loss": 0.0, "num_tokens": 4951714.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 886 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022001736326429367, "grad_norm": 0.0, "learning_rate": 5.57e-07, "loss": 0.0, "num_tokens": 4952620.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 887 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022026540989706064, "grad_norm": 0.0, "learning_rate": 5.565e-07, "loss": 0.0, "num_tokens": 4953500.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 888 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02205134565298276, "grad_norm": 0.0, "learning_rate": 5.560000000000001e-07, "loss": 0.0, "num_tokens": 4954436.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 911.0, "completions/max_terminated_length": 911.0, "completions/mean_length": 879.0, "completions/mean_terminated_length": 879.0, "completions/min_length": 847.0, "completions/min_terminated_length": 847.0, "epoch": 0.022076150316259455, "grad_norm": 0.0, "learning_rate": 5.555e-07, "loss": 0.0, "num_tokens": 4957054.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 819.0, "completions/max_terminated_length": 819.0, "completions/mean_length": 813.0, "completions/mean_terminated_length": 813.0, "completions/min_length": 807.0, "completions/min_terminated_length": 807.0, "epoch": 0.022100954979536153, "grad_norm": 0.0, "learning_rate": 5.55e-07, "loss": 0.0, "num_tokens": 4959514.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 891 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1255.0, "completions/max_terminated_length": 1255.0, "completions/mean_length": 918.0, "completions/mean_terminated_length": 918.0, "completions/min_length": 581.0, "completions/min_terminated_length": 581.0, "epoch": 0.02212575964281285, "grad_norm": 0.0, "learning_rate": 5.544999999999999e-07, "loss": 0.0, "num_tokens": 4962198.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 892 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6020.0, "completions/mean_length": 7106.0, "completions/mean_terminated_length": 6020.0, "completions/min_length": 6020.0, "completions/min_terminated_length": 6020.0, "epoch": 0.022150564306089544, "grad_norm": 2.8535373210906982, "learning_rate": 5.54e-07, "loss": -0.707, "num_tokens": 4969112.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 893 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02217536896936624, "grad_norm": 0.0, "learning_rate": 5.535e-07, "loss": 0.0, "num_tokens": 4969960.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 894 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022200173632642938, "grad_norm": 0.0, "learning_rate": 5.53e-07, "loss": 0.0, "num_tokens": 4970926.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1090.0, "completions/max_terminated_length": 1090.0, "completions/mean_length": 937.5, "completions/mean_terminated_length": 937.5, "completions/min_length": 785.0, "completions/min_terminated_length": 785.0, "epoch": 0.022224978295919632, "grad_norm": 0.0, "learning_rate": 5.525e-07, "loss": 0.0, "num_tokens": 4973595.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6429.0, "completions/mean_length": 7310.5, "completions/mean_terminated_length": 6429.0, "completions/min_length": 6429.0, "completions/min_terminated_length": 6429.0, "epoch": 0.02224978295919633, "grad_norm": 0.0, "learning_rate": 5.520000000000001e-07, "loss": 0.0, "num_tokens": 4980946.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 897 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022274587622473026, "grad_norm": 0.0, "learning_rate": 5.514999999999999e-07, "loss": 0.0, "num_tokens": 4981944.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 898 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4091.0, "completions/max_terminated_length": 4091.0, "completions/mean_length": 3115.5, "completions/mean_terminated_length": 3115.5, "completions/min_length": 2140.0, "completions/min_terminated_length": 2140.0, "epoch": 0.02229939228574972, "grad_norm": 0.0, "learning_rate": 5.51e-07, "loss": 0.0, "num_tokens": 4989175.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7657.0, "completions/max_terminated_length": 7657.0, "completions/mean_length": 6540.5, "completions/mean_terminated_length": 6540.5, "completions/min_length": 5424.0, "completions/min_terminated_length": 5424.0, "epoch": 0.022324196949026417, "grad_norm": 0.0, "learning_rate": 5.505e-07, "loss": 0.0, "num_tokens": 5003178.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7624.0, "completions/max_terminated_length": 7624.0, "completions/mean_length": 7355.5, "completions/mean_terminated_length": 7355.5, "completions/min_length": 7087.0, "completions/min_terminated_length": 7087.0, "epoch": 0.02234900161230311, "grad_norm": 2.3873422145843506, "learning_rate": 5.5e-07, "loss": -0.0258, "num_tokens": 5018777.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5027.0, "completions/mean_length": 6609.5, "completions/mean_terminated_length": 5027.0, "completions/min_length": 5027.0, "completions/min_terminated_length": 5027.0, "epoch": 0.02237380627557981, "grad_norm": 0.0, "learning_rate": 5.495e-07, "loss": 0.0, "num_tokens": 5024694.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7025.0, "completions/max_terminated_length": 7025.0, "completions/mean_length": 6238.5, "completions/mean_terminated_length": 6238.5, "completions/min_length": 5452.0, "completions/min_terminated_length": 5452.0, "epoch": 0.022398610938856506, "grad_norm": 0.0, "learning_rate": 5.490000000000001e-07, "loss": 0.0, "num_tokens": 5038269.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 903 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0224234156021332, "grad_norm": 0.0, "learning_rate": 5.484999999999999e-07, "loss": 0.0, "num_tokens": 5039163.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 904 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022448220265409897, "grad_norm": 0.0, "learning_rate": 5.48e-07, "loss": 0.0, "num_tokens": 5040135.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1602.0, "completions/max_terminated_length": 1602.0, "completions/mean_length": 1258.5, "completions/mean_terminated_length": 1258.5, "completions/min_length": 915.0, "completions/min_terminated_length": 915.0, "epoch": 0.022473024928686594, "grad_norm": 0.0, "learning_rate": 5.474999999999999e-07, "loss": 0.0, "num_tokens": 5043504.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1697.0, "completions/mean_length": 4944.5, "completions/mean_terminated_length": 1697.0, "completions/min_length": 1697.0, "completions/min_terminated_length": 1697.0, "epoch": 0.022497829591963288, "grad_norm": 0.0, "learning_rate": 5.47e-07, "loss": 0.0, "num_tokens": 5046083.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3587.0, "completions/max_terminated_length": 3587.0, "completions/mean_length": 2694.0, "completions/mean_terminated_length": 2694.0, "completions/min_length": 1801.0, "completions/min_terminated_length": 1801.0, "epoch": 0.022522634255239985, "grad_norm": 0.0, "learning_rate": 5.465e-07, "loss": 0.0, "num_tokens": 5052369.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4100.0, "completions/mean_length": 6146.0, "completions/mean_terminated_length": 4100.0, "completions/min_length": 4100.0, "completions/min_terminated_length": 4100.0, "epoch": 0.022547438918516682, "grad_norm": 4.32429313659668, "learning_rate": 5.46e-07, "loss": -0.707, "num_tokens": 5057379.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 909 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022572243581793376, "grad_norm": 0.0, "learning_rate": 5.455e-07, "loss": 0.0, "num_tokens": 5058313.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1721.0, "completions/mean_length": 4956.5, "completions/mean_terminated_length": 1721.0, "completions/min_length": 1721.0, "completions/min_terminated_length": 1721.0, "epoch": 0.022597048245070073, "grad_norm": 0.0, "learning_rate": 5.45e-07, "loss": 0.0, "num_tokens": 5060890.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 911 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02262185290834677, "grad_norm": 0.0, "learning_rate": 5.444999999999999e-07, "loss": 0.0, "num_tokens": 5061950.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1222.0, "completions/max_terminated_length": 1222.0, "completions/mean_length": 919.0, "completions/mean_terminated_length": 919.0, "completions/min_length": 616.0, "completions/min_terminated_length": 616.0, "epoch": 0.022646657571623464, "grad_norm": 0.0, "learning_rate": 5.44e-07, "loss": 0.0, "num_tokens": 5064628.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4935.0, "completions/max_terminated_length": 4935.0, "completions/mean_length": 4056.0, "completions/mean_terminated_length": 4056.0, "completions/min_length": 3177.0, "completions/min_terminated_length": 3177.0, "epoch": 0.02267146223490016, "grad_norm": 2.3403847217559814, "learning_rate": 5.435e-07, "loss": 0.1532, "num_tokens": 5073714.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 914 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02269626689817686, "grad_norm": 0.0, "learning_rate": 5.43e-07, "loss": 0.0, "num_tokens": 5074780.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7809.0, "completions/mean_length": 8000.5, "completions/mean_terminated_length": 7809.0, "completions/min_length": 7809.0, "completions/min_terminated_length": 7809.0, "epoch": 0.022721071561453553, "grad_norm": 0.0, "learning_rate": 5.425e-07, "loss": 0.0, "num_tokens": 5083549.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4334.0, "completions/max_terminated_length": 4334.0, "completions/mean_length": 4086.0, "completions/mean_terminated_length": 4086.0, "completions/min_length": 3838.0, "completions/min_terminated_length": 3838.0, "epoch": 0.02274587622473025, "grad_norm": 0.0, "learning_rate": 5.420000000000001e-07, "loss": 0.0, "num_tokens": 5092679.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 917 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022770680888006944, "grad_norm": 0.0, "learning_rate": 5.414999999999999e-07, "loss": 0.0, "num_tokens": 5093599.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3211.0, "completions/max_terminated_length": 3211.0, "completions/mean_length": 2552.0, "completions/mean_terminated_length": 2552.0, "completions/min_length": 1893.0, "completions/min_terminated_length": 1893.0, "epoch": 0.02279548555128364, "grad_norm": 3.860680103302002, "learning_rate": 5.41e-07, "loss": -0.1826, "num_tokens": 5099547.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 919 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3007.0, "completions/max_terminated_length": 3007.0, "completions/mean_length": 2563.0, "completions/mean_terminated_length": 2563.0, "completions/min_length": 2119.0, "completions/min_terminated_length": 2119.0, "epoch": 0.02282029021456034, "grad_norm": 0.0, "learning_rate": 5.405e-07, "loss": 0.0, "num_tokens": 5105631.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 920 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022845094877837032, "grad_norm": 0.0, "learning_rate": 5.4e-07, "loss": 0.0, "num_tokens": 5106531.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 921 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02286989954111373, "grad_norm": 0.0, "learning_rate": 5.395e-07, "loss": 0.0, "num_tokens": 5107499.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 922 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6032.0, "completions/mean_length": 7112.0, "completions/mean_terminated_length": 6032.0, "completions/min_length": 6032.0, "completions/min_terminated_length": 6032.0, "epoch": 0.022894704204390427, "grad_norm": 2.859607696533203, "learning_rate": 5.39e-07, "loss": -0.707, "num_tokens": 5114365.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 923 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02291950886766712, "grad_norm": 0.0, "learning_rate": 5.384999999999999e-07, "loss": 0.0, "num_tokens": 5115229.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 924 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.022944313530943818, "grad_norm": 0.0, "learning_rate": 5.38e-07, "loss": 0.0, "num_tokens": 5116105.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4038.0, "completions/mean_length": 6115.0, "completions/mean_terminated_length": 4038.0, "completions/min_length": 4038.0, "completions/min_terminated_length": 4038.0, "epoch": 0.022969118194220515, "grad_norm": 0.0, "learning_rate": 5.374999999999999e-07, "loss": 0.0, "num_tokens": 5120985.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8000.0, "completions/max_terminated_length": 8000.0, "completions/mean_length": 6548.5, "completions/mean_terminated_length": 6548.5, "completions/min_length": 5097.0, "completions/min_terminated_length": 5097.0, "epoch": 0.02299392285749721, "grad_norm": 0.0, "learning_rate": 5.37e-07, "loss": 0.0, "num_tokens": 5135196.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 927 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023018727520773906, "grad_norm": 0.0, "learning_rate": 5.365e-07, "loss": 0.0, "num_tokens": 5136072.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 928 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023043532184050603, "grad_norm": 0.0, "learning_rate": 5.36e-07, "loss": 0.0, "num_tokens": 5136982.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 929 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023068336847327297, "grad_norm": 0.0, "learning_rate": 5.355e-07, "loss": 0.0, "num_tokens": 5138006.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2547.0, "completions/max_terminated_length": 2547.0, "completions/mean_length": 2444.0, "completions/mean_terminated_length": 2444.0, "completions/min_length": 2341.0, "completions/min_terminated_length": 2341.0, "epoch": 0.023093141510603994, "grad_norm": 0.0, "learning_rate": 5.35e-07, "loss": 0.0, "num_tokens": 5143818.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7557.0, "completions/max_terminated_length": 7557.0, "completions/mean_length": 7475.5, "completions/mean_terminated_length": 7475.5, "completions/min_length": 7394.0, "completions/min_terminated_length": 7394.0, "epoch": 0.023117946173880688, "grad_norm": 0.0, "learning_rate": 5.344999999999999e-07, "loss": 0.0, "num_tokens": 5159643.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 932 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2069.0, "completions/max_terminated_length": 2069.0, "completions/mean_length": 1554.5, "completions/mean_terminated_length": 1554.5, "completions/min_length": 1040.0, "completions/min_terminated_length": 1040.0, "epoch": 0.023142750837157385, "grad_norm": 0.0, "learning_rate": 5.34e-07, "loss": 0.0, "num_tokens": 5164800.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 933 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023167555500434082, "grad_norm": 0.0, "learning_rate": 5.335e-07, "loss": 0.0, "num_tokens": 5165724.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2134.0, "completions/max_terminated_length": 2134.0, "completions/mean_length": 1545.0, "completions/mean_terminated_length": 1545.0, "completions/min_length": 956.0, "completions/min_terminated_length": 956.0, "epoch": 0.023192360163710776, "grad_norm": 0.0, "learning_rate": 5.33e-07, "loss": 0.0, "num_tokens": 5169656.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 978.0, "completions/max_terminated_length": 978.0, "completions/mean_length": 831.0, "completions/mean_terminated_length": 831.0, "completions/min_length": 684.0, "completions/min_terminated_length": 684.0, "epoch": 0.023217164826987473, "grad_norm": 7.011067867279053, "learning_rate": 5.325e-07, "loss": -0.1251, "num_tokens": 5172224.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2851.0, "completions/max_terminated_length": 2851.0, "completions/mean_length": 1911.0, "completions/mean_terminated_length": 1911.0, "completions/min_length": 971.0, "completions/min_terminated_length": 971.0, "epoch": 0.02324196949026417, "grad_norm": 0.0, "learning_rate": 5.32e-07, "loss": 0.0, "num_tokens": 5176936.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 937 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023266774153540865, "grad_norm": 0.0, "learning_rate": 5.314999999999999e-07, "loss": 0.0, "num_tokens": 5177820.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 938 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023291578816817562, "grad_norm": 0.0, "learning_rate": 5.31e-07, "loss": 0.0, "num_tokens": 5178708.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6227.0, "completions/mean_length": 7209.5, "completions/mean_terminated_length": 6227.0, "completions/min_length": 6227.0, "completions/min_terminated_length": 6227.0, "epoch": 0.02331638348009426, "grad_norm": 3.2511298656463623, "learning_rate": 5.304999999999999e-07, "loss": -0.707, "num_tokens": 5185857.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 940 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6294.0, "completions/mean_length": 7243.0, "completions/mean_terminated_length": 6294.0, "completions/min_length": 6294.0, "completions/min_terminated_length": 6294.0, "epoch": 0.023341188143370953, "grad_norm": 0.0, "learning_rate": 5.3e-07, "loss": 0.0, "num_tokens": 5193075.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 941 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02336599280664765, "grad_norm": 0.0, "learning_rate": 5.295e-07, "loss": 0.0, "num_tokens": 5194111.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 942 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023390797469924347, "grad_norm": 0.0, "learning_rate": 5.29e-07, "loss": 0.0, "num_tokens": 5195033.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 946.0, "completions/mean_length": 4569.0, "completions/mean_terminated_length": 946.0, "completions/min_length": 946.0, "completions/min_terminated_length": 946.0, "epoch": 0.02341560213320104, "grad_norm": 8.350257873535156, "learning_rate": 5.284999999999999e-07, "loss": -0.707, "num_tokens": 5196949.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1022.0, "completions/max_terminated_length": 1022.0, "completions/mean_length": 819.5, "completions/mean_terminated_length": 819.5, "completions/min_length": 617.0, "completions/min_terminated_length": 617.0, "epoch": 0.02344040679647774, "grad_norm": 0.0, "learning_rate": 5.28e-07, "loss": 0.0, "num_tokens": 5199432.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2102.0, "completions/max_terminated_length": 2102.0, "completions/mean_length": 2017.5, "completions/mean_terminated_length": 2017.5, "completions/min_length": 1933.0, "completions/min_terminated_length": 1933.0, "epoch": 0.023465211459754432, "grad_norm": 0.0, "learning_rate": 5.274999999999999e-07, "loss": 0.0, "num_tokens": 5204415.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2382.0, "completions/max_terminated_length": 2382.0, "completions/mean_length": 2293.5, "completions/mean_terminated_length": 2293.5, "completions/min_length": 2205.0, "completions/min_terminated_length": 2205.0, "epoch": 0.02349001612303113, "grad_norm": 0.0, "learning_rate": 5.27e-07, "loss": 0.0, "num_tokens": 5210052.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 947 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1599.0, "completions/max_terminated_length": 1599.0, "completions/mean_length": 1232.0, "completions/mean_terminated_length": 1232.0, "completions/min_length": 865.0, "completions/min_terminated_length": 865.0, "epoch": 0.023514820786307827, "grad_norm": 0.0, "learning_rate": 5.265e-07, "loss": 0.0, "num_tokens": 5213372.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 948 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02353962544958452, "grad_norm": 0.0, "learning_rate": 5.26e-07, "loss": 0.0, "num_tokens": 5214186.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5488.0, "completions/max_terminated_length": 5488.0, "completions/mean_length": 4318.0, "completions/mean_terminated_length": 4318.0, "completions/min_length": 3148.0, "completions/min_terminated_length": 3148.0, "epoch": 0.023564430112861218, "grad_norm": 0.0, "learning_rate": 5.255e-07, "loss": 0.0, "num_tokens": 5223680.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 950 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023589234776137915, "grad_norm": 0.0, "learning_rate": 5.25e-07, "loss": 0.0, "num_tokens": 5224534.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5449.0, "completions/mean_length": 6820.5, "completions/mean_terminated_length": 5449.0, "completions/min_length": 5449.0, "completions/min_terminated_length": 5449.0, "epoch": 0.02361403943941461, "grad_norm": 0.0, "learning_rate": 5.244999999999999e-07, "loss": 0.0, "num_tokens": 5230819.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6693.0, "completions/max_terminated_length": 6693.0, "completions/mean_length": 6062.5, "completions/mean_terminated_length": 6062.5, "completions/min_length": 5432.0, "completions/min_terminated_length": 5432.0, "epoch": 0.023638844102691306, "grad_norm": 0.0, "learning_rate": 5.24e-07, "loss": 0.0, "num_tokens": 5243778.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2765.0, "completions/max_terminated_length": 2765.0, "completions/mean_length": 2042.5, "completions/mean_terminated_length": 2042.5, "completions/min_length": 1320.0, "completions/min_terminated_length": 1320.0, "epoch": 0.023663648765968003, "grad_norm": 4.723125457763672, "learning_rate": 5.234999999999999e-07, "loss": -0.2501, "num_tokens": 5248905.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 954 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023688453429244697, "grad_norm": 0.0, "learning_rate": 5.23e-07, "loss": 0.0, "num_tokens": 5249849.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 955 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023713258092521394, "grad_norm": 0.0, "learning_rate": 5.225e-07, "loss": 0.0, "num_tokens": 5250795.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1809.0, "completions/max_terminated_length": 1809.0, "completions/mean_length": 1472.0, "completions/mean_terminated_length": 1472.0, "completions/min_length": 1135.0, "completions/min_terminated_length": 1135.0, "epoch": 0.02373806275579809, "grad_norm": 0.0, "learning_rate": 5.22e-07, "loss": 0.0, "num_tokens": 5254567.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2601.0, "completions/max_terminated_length": 2601.0, "completions/mean_length": 1864.5, "completions/mean_terminated_length": 1864.5, "completions/min_length": 1128.0, "completions/min_terminated_length": 1128.0, "epoch": 0.023762867419074785, "grad_norm": 0.0, "learning_rate": 5.214999999999999e-07, "loss": 0.0, "num_tokens": 5259126.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 744.0, "completions/max_terminated_length": 744.0, "completions/mean_length": 568.5, "completions/mean_terminated_length": 568.5, "completions/min_length": 393.0, "completions/min_terminated_length": 393.0, "epoch": 0.023787672082351483, "grad_norm": 0.0, "learning_rate": 5.21e-07, "loss": 0.0, "num_tokens": 5261101.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4959.0, "completions/mean_length": 6575.5, "completions/mean_terminated_length": 4959.0, "completions/min_length": 4959.0, "completions/min_terminated_length": 4959.0, "epoch": 0.02381247674562818, "grad_norm": 3.716373920440674, "learning_rate": 5.204999999999999e-07, "loss": -0.707, "num_tokens": 5266918.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 960 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.023837281408904874, "grad_norm": 0.0, "learning_rate": 5.2e-07, "loss": 0.0, "num_tokens": 5267754.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1692.0, "completions/max_terminated_length": 1692.0, "completions/mean_length": 1313.0, "completions/mean_terminated_length": 1313.0, "completions/min_length": 934.0, "completions/min_terminated_length": 934.0, "epoch": 0.02386208607218157, "grad_norm": 0.0, "learning_rate": 5.195e-07, "loss": 0.0, "num_tokens": 5271202.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5397.0, "completions/max_terminated_length": 5397.0, "completions/mean_length": 4261.0, "completions/mean_terminated_length": 4261.0, "completions/min_length": 3125.0, "completions/min_terminated_length": 3125.0, "epoch": 0.023886890735458265, "grad_norm": 0.0, "learning_rate": 5.19e-07, "loss": 0.0, "num_tokens": 5280642.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2946.0, "completions/max_terminated_length": 2946.0, "completions/mean_length": 2343.5, "completions/mean_terminated_length": 2343.5, "completions/min_length": 1741.0, "completions/min_terminated_length": 1741.0, "epoch": 0.023911695398734962, "grad_norm": 0.0, "learning_rate": 5.184999999999999e-07, "loss": 0.0, "num_tokens": 5286255.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 240.0, "completions/mean_length": 4216.0, "completions/mean_terminated_length": 240.0, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "epoch": 0.02393650006201166, "grad_norm": 0.0, "learning_rate": 5.18e-07, "loss": 0.0, "num_tokens": 5287381.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1223.0, "completions/max_terminated_length": 1223.0, "completions/mean_length": 1009.0, "completions/mean_terminated_length": 1009.0, "completions/min_length": 795.0, "completions/min_terminated_length": 795.0, "epoch": 0.023961304725288353, "grad_norm": 0.0, "learning_rate": 5.174999999999999e-07, "loss": 0.0, "num_tokens": 5290265.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 966 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02398610938856505, "grad_norm": 0.0, "learning_rate": 5.17e-07, "loss": 0.0, "num_tokens": 5291153.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 967 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5735.0, "completions/max_terminated_length": 5735.0, "completions/mean_length": 4998.5, "completions/mean_terminated_length": 4998.5, "completions/min_length": 4262.0, "completions/min_terminated_length": 4262.0, "epoch": 0.024010914051841747, "grad_norm": 0.0, "learning_rate": 5.164999999999999e-07, "loss": 0.0, "num_tokens": 5302036.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 968 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02403571871511844, "grad_norm": 0.0, "learning_rate": 5.16e-07, "loss": 0.0, "num_tokens": 5303016.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 969 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02406052337839514, "grad_norm": 0.0, "learning_rate": 5.155e-07, "loss": 0.0, "num_tokens": 5303928.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 970 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024085328041671836, "grad_norm": 0.0, "learning_rate": 5.149999999999999e-07, "loss": 0.0, "num_tokens": 5304822.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 971 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02411013270494853, "grad_norm": 0.0, "learning_rate": 5.144999999999999e-07, "loss": 0.0, "num_tokens": 5305788.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3035.0, "completions/max_terminated_length": 3035.0, "completions/mean_length": 2026.0, "completions/mean_terminated_length": 2026.0, "completions/min_length": 1017.0, "completions/min_terminated_length": 1017.0, "epoch": 0.024134937368225227, "grad_norm": 4.131929874420166, "learning_rate": 5.14e-07, "loss": 0.3521, "num_tokens": 5310726.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 973 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024159742031501924, "grad_norm": 0.0, "learning_rate": 5.134999999999999e-07, "loss": 0.0, "num_tokens": 5311694.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1088.0, "completions/max_terminated_length": 1088.0, "completions/mean_length": 973.0, "completions/mean_terminated_length": 973.0, "completions/min_length": 858.0, "completions/min_terminated_length": 858.0, "epoch": 0.024184546694778618, "grad_norm": 0.0, "learning_rate": 5.13e-07, "loss": 0.0, "num_tokens": 5314448.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4855.0, "completions/mean_length": 6523.5, "completions/mean_terminated_length": 4855.0, "completions/min_length": 4855.0, "completions/min_terminated_length": 4855.0, "epoch": 0.024209351358055315, "grad_norm": 3.3595023155212402, "learning_rate": 5.125e-07, "loss": -0.707, "num_tokens": 5320239.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 976 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02423415602133201, "grad_norm": 0.0, "learning_rate": 5.12e-07, "loss": 0.0, "num_tokens": 5321125.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 977 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5357.0, "completions/max_terminated_length": 5357.0, "completions/mean_length": 5102.5, "completions/mean_terminated_length": 5102.5, "completions/min_length": 4848.0, "completions/min_terminated_length": 4848.0, "epoch": 0.024258960684608706, "grad_norm": 2.53116512298584, "learning_rate": 5.114999999999999e-07, "loss": -0.0353, "num_tokens": 5332178.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8094.0, "completions/mean_length": 8143.0, "completions/mean_terminated_length": 8094.0, "completions/min_length": 8094.0, "completions/min_terminated_length": 8094.0, "epoch": 0.024283765347885403, "grad_norm": 3.1025428771972656, "learning_rate": 5.11e-07, "loss": -0.707, "num_tokens": 5341186.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 979 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2353.0, "completions/max_terminated_length": 2353.0, "completions/mean_length": 2070.0, "completions/mean_terminated_length": 2070.0, "completions/min_length": 1787.0, "completions/min_terminated_length": 1787.0, "epoch": 0.024308570011162097, "grad_norm": 0.0, "learning_rate": 5.104999999999999e-07, "loss": 0.0, "num_tokens": 5346130.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 980 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024333374674438794, "grad_norm": 0.0, "learning_rate": 5.1e-07, "loss": 0.0, "num_tokens": 5347016.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4938.0, "completions/max_terminated_length": 4938.0, "completions/mean_length": 4110.0, "completions/mean_terminated_length": 4110.0, "completions/min_length": 3282.0, "completions/min_terminated_length": 3282.0, "epoch": 0.02435817933771549, "grad_norm": 0.0, "learning_rate": 5.095e-07, "loss": 0.0, "num_tokens": 5356098.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7388.0, "completions/mean_length": 7790.0, "completions/mean_terminated_length": 7388.0, "completions/min_length": 7388.0, "completions/min_terminated_length": 7388.0, "epoch": 0.024382984000992185, "grad_norm": 0.0, "learning_rate": 5.09e-07, "loss": 0.0, "num_tokens": 5364412.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 983 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024407788664268883, "grad_norm": 0.0, "learning_rate": 5.085e-07, "loss": 0.0, "num_tokens": 5365314.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1479.0, "completions/max_terminated_length": 1479.0, "completions/mean_length": 1284.5, "completions/mean_terminated_length": 1284.5, "completions/min_length": 1090.0, "completions/min_terminated_length": 1090.0, "epoch": 0.02443259332754558, "grad_norm": 0.0, "learning_rate": 5.079999999999999e-07, "loss": 0.0, "num_tokens": 5368705.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 985 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024457397990822274, "grad_norm": 0.0, "learning_rate": 5.074999999999999e-07, "loss": 0.0, "num_tokens": 5369631.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 931.0, "completions/max_terminated_length": 931.0, "completions/mean_length": 600.0, "completions/mean_terminated_length": 600.0, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "epoch": 0.02448220265409897, "grad_norm": 0.0, "learning_rate": 5.07e-07, "loss": 0.0, "num_tokens": 5371621.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5074.0, "completions/mean_length": 6633.0, "completions/mean_terminated_length": 5074.0, "completions/min_length": 5074.0, "completions/min_terminated_length": 5074.0, "epoch": 0.024507007317375668, "grad_norm": 3.435997486114502, "learning_rate": 5.064999999999999e-07, "loss": -0.707, "num_tokens": 5377591.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 988 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024531811980652362, "grad_norm": 0.0, "learning_rate": 5.06e-07, "loss": 0.0, "num_tokens": 5378519.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 921.0, "completions/max_terminated_length": 921.0, "completions/mean_length": 719.0, "completions/mean_terminated_length": 719.0, "completions/min_length": 517.0, "completions/min_terminated_length": 517.0, "epoch": 0.02455661664392906, "grad_norm": 0.0, "learning_rate": 5.055e-07, "loss": 0.0, "num_tokens": 5380771.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 990 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7517.0, "completions/max_terminated_length": 7517.0, "completions/mean_length": 7355.0, "completions/mean_terminated_length": 7355.0, "completions/min_length": 7193.0, "completions/min_terminated_length": 7193.0, "epoch": 0.024581421307205753, "grad_norm": 2.380869150161743, "learning_rate": 5.049999999999999e-07, "loss": 0.0156, "num_tokens": 5396515.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 991 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02460622597048245, "grad_norm": 0.0, "learning_rate": 5.044999999999999e-07, "loss": 0.0, "num_tokens": 5397439.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6047.0, "completions/max_terminated_length": 6047.0, "completions/mean_length": 5765.5, "completions/mean_terminated_length": 5765.5, "completions/min_length": 5484.0, "completions/min_terminated_length": 5484.0, "epoch": 0.024631030633759148, "grad_norm": 0.0, "learning_rate": 5.04e-07, "loss": 0.0, "num_tokens": 5410028.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 993 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02465583529703584, "grad_norm": 0.0, "learning_rate": 5.034999999999999e-07, "loss": 0.0, "num_tokens": 5410952.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1246.0, "completions/max_terminated_length": 1246.0, "completions/mean_length": 1090.0, "completions/mean_terminated_length": 1090.0, "completions/min_length": 934.0, "completions/min_terminated_length": 934.0, "epoch": 0.02468063996031254, "grad_norm": 0.0, "learning_rate": 5.03e-07, "loss": 0.0, "num_tokens": 5413994.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4133.0, "completions/mean_length": 6162.5, "completions/mean_terminated_length": 4133.0, "completions/min_length": 4133.0, "completions/min_terminated_length": 4133.0, "epoch": 0.024705444623589236, "grad_norm": 4.530544281005859, "learning_rate": 5.025e-07, "loss": -0.707, "num_tokens": 5419251.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1569.0, "completions/max_terminated_length": 1569.0, "completions/mean_length": 1441.5, "completions/mean_terminated_length": 1441.5, "completions/min_length": 1314.0, "completions/min_terminated_length": 1314.0, "epoch": 0.02473024928686593, "grad_norm": 0.0, "learning_rate": 5.02e-07, "loss": 0.0, "num_tokens": 5423028.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 997 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.024755053950142627, "grad_norm": 0.0, "learning_rate": 5.014999999999999e-07, "loss": 0.0, "num_tokens": 5423892.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6905.0, "completions/max_terminated_length": 6905.0, "completions/mean_length": 6898.5, "completions/mean_terminated_length": 6898.5, "completions/min_length": 6892.0, "completions/min_terminated_length": 6892.0, "epoch": 0.024779858613419324, "grad_norm": 0.0, "learning_rate": 5.009999999999999e-07, "loss": 0.0, "num_tokens": 5438557.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4101.0, "completions/max_terminated_length": 4101.0, "completions/mean_length": 3884.5, "completions/mean_terminated_length": 3884.5, "completions/min_length": 3668.0, "completions/min_terminated_length": 3668.0, "epoch": 0.024804663276696018, "grad_norm": 0.0, "learning_rate": 5.004999999999999e-07, "loss": 0.0, "num_tokens": 5447266.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1000 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2666.0, "completions/max_terminated_length": 2666.0, "completions/mean_length": 2313.0, "completions/mean_terminated_length": 2313.0, "completions/min_length": 1960.0, "completions/min_terminated_length": 1960.0, "epoch": 0.024829467939972715, "grad_norm": 0.0, "learning_rate": 5e-07, "loss": 0.0, "num_tokens": 5452796.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3761.0, "completions/max_terminated_length": 3761.0, "completions/mean_length": 3429.5, "completions/mean_terminated_length": 3429.5, "completions/min_length": 3098.0, "completions/min_terminated_length": 3098.0, "epoch": 0.024854272603249412, "grad_norm": 0.0, "learning_rate": 4.994999999999999e-07, "loss": 0.0, "num_tokens": 5460521.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2680.0, "completions/max_terminated_length": 2680.0, "completions/mean_length": 2049.0, "completions/mean_terminated_length": 2049.0, "completions/min_length": 1418.0, "completions/min_terminated_length": 1418.0, "epoch": 0.024879077266526106, "grad_norm": 0.0, "learning_rate": 4.99e-07, "loss": 0.0, "num_tokens": 5465461.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5542.0, "completions/mean_length": 6867.0, "completions/mean_terminated_length": 5542.0, "completions/min_length": 5542.0, "completions/min_terminated_length": 5542.0, "epoch": 0.024903881929802803, "grad_norm": 0.0, "learning_rate": 4.985e-07, "loss": 0.0, "num_tokens": 5472001.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7880.0, "completions/max_terminated_length": 7880.0, "completions/mean_length": 5909.5, "completions/mean_terminated_length": 5909.5, "completions/min_length": 3939.0, "completions/min_terminated_length": 3939.0, "epoch": 0.024928686593079497, "grad_norm": 0.0, "learning_rate": 4.979999999999999e-07, "loss": 0.0, "num_tokens": 5484686.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1326.0, "completions/max_terminated_length": 1326.0, "completions/mean_length": 1204.0, "completions/mean_terminated_length": 1204.0, "completions/min_length": 1082.0, "completions/min_terminated_length": 1082.0, "epoch": 0.024953491256356194, "grad_norm": 0.0, "learning_rate": 4.975e-07, "loss": 0.0, "num_tokens": 5487930.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2460.0, "completions/max_terminated_length": 2460.0, "completions/mean_length": 2100.5, "completions/mean_terminated_length": 2100.5, "completions/min_length": 1741.0, "completions/min_terminated_length": 1741.0, "epoch": 0.02497829591963289, "grad_norm": 0.0, "learning_rate": 4.97e-07, "loss": 0.0, "num_tokens": 5492979.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1007 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025003100582909586, "grad_norm": 0.0, "learning_rate": 4.964999999999999e-07, "loss": 0.0, "num_tokens": 5493941.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6679.0, "completions/max_terminated_length": 6679.0, "completions/mean_length": 6104.5, "completions/mean_terminated_length": 6104.5, "completions/min_length": 5530.0, "completions/min_terminated_length": 5530.0, "epoch": 0.025027905246186283, "grad_norm": 2.285290479660034, "learning_rate": 4.96e-07, "loss": -0.0665, "num_tokens": 5507038.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7378.0, "completions/mean_length": 7785.0, "completions/mean_terminated_length": 7378.0, "completions/min_length": 7378.0, "completions/min_terminated_length": 7378.0, "epoch": 0.02505270990946298, "grad_norm": 0.0, "learning_rate": 4.955e-07, "loss": 0.0, "num_tokens": 5515374.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1010 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025077514572739674, "grad_norm": 0.0, "learning_rate": 4.95e-07, "loss": 0.0, "num_tokens": 5516416.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1011 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7550.0, "completions/mean_length": 7871.0, "completions/mean_terminated_length": 7550.0, "completions/min_length": 7550.0, "completions/min_terminated_length": 7550.0, "epoch": 0.02510231923601637, "grad_norm": 0.0, "learning_rate": 4.945e-07, "loss": 0.0, "num_tokens": 5525460.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1012 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7883.0, "completions/max_terminated_length": 7883.0, "completions/mean_length": 6175.5, "completions/mean_terminated_length": 6175.5, "completions/min_length": 4468.0, "completions/min_terminated_length": 4468.0, "epoch": 0.02512712389929307, "grad_norm": 0.0, "learning_rate": 4.94e-07, "loss": 0.0, "num_tokens": 5538653.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1134.0, "completions/max_terminated_length": 1134.0, "completions/mean_length": 934.5, "completions/mean_terminated_length": 934.5, "completions/min_length": 735.0, "completions/min_terminated_length": 735.0, "epoch": 0.025151928562569762, "grad_norm": 0.0, "learning_rate": 4.935e-07, "loss": 0.0, "num_tokens": 5541366.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1014 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02517673322584646, "grad_norm": 0.0, "learning_rate": 4.93e-07, "loss": 0.0, "num_tokens": 5542308.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1015 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025201537889123157, "grad_norm": 0.0, "learning_rate": 4.924999999999999e-07, "loss": 0.0, "num_tokens": 5543416.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1016 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02522634255239985, "grad_norm": 0.0, "learning_rate": 4.92e-07, "loss": 0.0, "num_tokens": 5544862.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1017 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025251147215676548, "grad_norm": 0.0, "learning_rate": 4.915e-07, "loss": 0.0, "num_tokens": 5545730.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1378.0, "completions/max_terminated_length": 1378.0, "completions/mean_length": 1357.5, "completions/mean_terminated_length": 1357.5, "completions/min_length": 1337.0, "completions/min_terminated_length": 1337.0, "epoch": 0.025275951878953245, "grad_norm": 0.0, "learning_rate": 4.909999999999999e-07, "loss": 0.0, "num_tokens": 5549261.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1019 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02530075654222994, "grad_norm": 0.0, "learning_rate": 4.905e-07, "loss": 0.0, "num_tokens": 5550139.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1020 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6727.0, "completions/max_terminated_length": 6727.0, "completions/mean_length": 4958.5, "completions/mean_terminated_length": 4958.5, "completions/min_length": 3190.0, "completions/min_terminated_length": 3190.0, "epoch": 0.025325561205506636, "grad_norm": 0.0, "learning_rate": 4.9e-07, "loss": 0.0, "num_tokens": 5560904.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1021 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02535036586878333, "grad_norm": 0.0, "learning_rate": 4.894999999999999e-07, "loss": 0.0, "num_tokens": 5561978.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1022 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5920.0, "completions/max_terminated_length": 5920.0, "completions/mean_length": 4643.0, "completions/mean_terminated_length": 4643.0, "completions/min_length": 3366.0, "completions/min_terminated_length": 3366.0, "epoch": 0.025375170532060027, "grad_norm": 0.0, "learning_rate": 4.89e-07, "loss": 0.0, "num_tokens": 5572088.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1023 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025399975195336724, "grad_norm": 0.0, "learning_rate": 4.885e-07, "loss": 0.0, "num_tokens": 5572968.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1024 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025424779858613418, "grad_norm": 0.0, "learning_rate": 4.879999999999999e-07, "loss": 0.0, "num_tokens": 5573956.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3092.0, "completions/max_terminated_length": 3092.0, "completions/mean_length": 2752.5, "completions/mean_terminated_length": 2752.5, "completions/min_length": 2413.0, "completions/min_terminated_length": 2413.0, "epoch": 0.025449584521890115, "grad_norm": 0.0, "learning_rate": 4.875e-07, "loss": 0.0, "num_tokens": 5580371.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1026 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025474389185166812, "grad_norm": 0.0, "learning_rate": 4.87e-07, "loss": 0.0, "num_tokens": 5581421.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3936.0, "completions/max_terminated_length": 3936.0, "completions/mean_length": 2886.5, "completions/mean_terminated_length": 2886.5, "completions/min_length": 1837.0, "completions/min_terminated_length": 1837.0, "epoch": 0.025499193848443506, "grad_norm": 0.0, "learning_rate": 4.864999999999999e-07, "loss": 0.0, "num_tokens": 5588144.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1028 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3120.0, "completions/max_terminated_length": 3120.0, "completions/mean_length": 2788.5, "completions/mean_terminated_length": 2788.5, "completions/min_length": 2457.0, "completions/min_terminated_length": 2457.0, "epoch": 0.025523998511720204, "grad_norm": 0.0, "learning_rate": 4.86e-07, "loss": 0.0, "num_tokens": 5594645.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1029 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0255488031749969, "grad_norm": 0.0, "learning_rate": 4.854999999999999e-07, "loss": 0.0, "num_tokens": 5595883.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1030 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5329.0, "completions/mean_length": 6760.5, "completions/mean_terminated_length": 5329.0, "completions/min_length": 5329.0, "completions/min_terminated_length": 5329.0, "epoch": 0.025573607838273595, "grad_norm": 0.0, "learning_rate": 4.85e-07, "loss": 0.0, "num_tokens": 5602028.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1031 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3876.0, "completions/max_terminated_length": 3876.0, "completions/mean_length": 3228.0, "completions/mean_terminated_length": 3228.0, "completions/min_length": 2580.0, "completions/min_terminated_length": 2580.0, "epoch": 0.025598412501550292, "grad_norm": 0.0, "learning_rate": 4.845e-07, "loss": 0.0, "num_tokens": 5609360.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1032 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02562321716482699, "grad_norm": 0.0, "learning_rate": 4.839999999999999e-07, "loss": 0.0, "num_tokens": 5610230.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5135.0, "completions/max_terminated_length": 5135.0, "completions/mean_length": 3960.0, "completions/mean_terminated_length": 3960.0, "completions/min_length": 2785.0, "completions/min_terminated_length": 2785.0, "epoch": 0.025648021828103683, "grad_norm": 0.0, "learning_rate": 4.835e-07, "loss": 0.0, "num_tokens": 5618990.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3611.0, "completions/max_terminated_length": 3611.0, "completions/mean_length": 3481.5, "completions/mean_terminated_length": 3481.5, "completions/min_length": 3352.0, "completions/min_terminated_length": 3352.0, "epoch": 0.02567282649138038, "grad_norm": 0.0, "learning_rate": 4.83e-07, "loss": 0.0, "num_tokens": 5627185.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8130.0, "completions/max_terminated_length": 8130.0, "completions/mean_length": 7721.5, "completions/mean_terminated_length": 7721.5, "completions/min_length": 7313.0, "completions/min_terminated_length": 7313.0, "epoch": 0.025697631154657074, "grad_norm": 2.4696340560913086, "learning_rate": 4.824999999999999e-07, "loss": 0.0374, "num_tokens": 5643420.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6553.0, "completions/max_terminated_length": 6553.0, "completions/mean_length": 4152.0, "completions/mean_terminated_length": 4152.0, "completions/min_length": 1751.0, "completions/min_terminated_length": 1751.0, "epoch": 0.02572243581793377, "grad_norm": 2.51590895652771, "learning_rate": 4.82e-07, "loss": 0.4088, "num_tokens": 5652686.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1067.0, "completions/max_terminated_length": 1067.0, "completions/mean_length": 1022.5, "completions/mean_terminated_length": 1022.5, "completions/min_length": 978.0, "completions/min_terminated_length": 978.0, "epoch": 0.02574724048121047, "grad_norm": 0.0, "learning_rate": 4.815e-07, "loss": 0.0, "num_tokens": 5655523.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1038 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7839.0, "completions/mean_length": 8015.5, "completions/mean_terminated_length": 7839.0, "completions/min_length": 7839.0, "completions/min_terminated_length": 7839.0, "epoch": 0.025772045144487162, "grad_norm": 3.1193325519561768, "learning_rate": 4.809999999999999e-07, "loss": -0.707, "num_tokens": 5664214.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7146.0, "completions/mean_length": 7669.0, "completions/mean_terminated_length": 7146.0, "completions/min_length": 7146.0, "completions/min_terminated_length": 7146.0, "epoch": 0.02579684980776386, "grad_norm": 3.317911386489868, "learning_rate": 4.805e-07, "loss": -0.707, "num_tokens": 5672216.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1040 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1440.0, "completions/max_terminated_length": 1440.0, "completions/mean_length": 1252.0, "completions/mean_terminated_length": 1252.0, "completions/min_length": 1064.0, "completions/min_terminated_length": 1064.0, "epoch": 0.025821654471040557, "grad_norm": 0.0, "learning_rate": 4.8e-07, "loss": 0.0, "num_tokens": 5675624.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7866.0, "completions/max_terminated_length": 7866.0, "completions/mean_length": 6524.0, "completions/mean_terminated_length": 6524.0, "completions/min_length": 5182.0, "completions/min_terminated_length": 5182.0, "epoch": 0.02584645913431725, "grad_norm": 0.0, "learning_rate": 4.794999999999999e-07, "loss": 0.0, "num_tokens": 5689716.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1042 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025871263797593948, "grad_norm": 0.0, "learning_rate": 4.79e-07, "loss": 0.0, "num_tokens": 5690644.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1043 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.025896068460870645, "grad_norm": 0.0, "learning_rate": 4.785e-07, "loss": 0.0, "num_tokens": 5691594.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1044 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02592087312414734, "grad_norm": 0.0, "learning_rate": 4.779999999999999e-07, "loss": 0.0, "num_tokens": 5692560.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1671.0, "completions/max_terminated_length": 1671.0, "completions/mean_length": 1463.0, "completions/mean_terminated_length": 1463.0, "completions/min_length": 1255.0, "completions/min_terminated_length": 1255.0, "epoch": 0.025945677787424036, "grad_norm": 0.0, "learning_rate": 4.775e-07, "loss": 0.0, "num_tokens": 5696334.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5031.0, "completions/mean_length": 6611.5, "completions/mean_terminated_length": 5031.0, "completions/min_length": 5031.0, "completions/min_terminated_length": 5031.0, "epoch": 0.025970482450700733, "grad_norm": 3.5271778106689453, "learning_rate": 4.769999999999999e-07, "loss": -0.707, "num_tokens": 5702189.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3889.0, "completions/max_terminated_length": 3889.0, "completions/mean_length": 3220.0, "completions/mean_terminated_length": 3220.0, "completions/min_length": 2551.0, "completions/min_terminated_length": 2551.0, "epoch": 0.025995287113977427, "grad_norm": 0.0, "learning_rate": 4.7649999999999996e-07, "loss": 0.0, "num_tokens": 5709495.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2892.0, "completions/mean_length": 5542.0, "completions/mean_terminated_length": 2892.0, "completions/min_length": 2892.0, "completions/min_terminated_length": 2892.0, "epoch": 0.026020091777254124, "grad_norm": 5.097677230834961, "learning_rate": 4.76e-07, "loss": -0.707, "num_tokens": 5713249.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1049 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3442.0, "completions/max_terminated_length": 3442.0, "completions/mean_length": 2549.5, "completions/mean_terminated_length": 2549.5, "completions/min_length": 1657.0, "completions/min_terminated_length": 1657.0, "epoch": 0.026044896440530818, "grad_norm": 0.0, "learning_rate": 4.7549999999999994e-07, "loss": 0.0, "num_tokens": 5719276.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1050 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5051.0, "completions/max_terminated_length": 5051.0, "completions/mean_length": 3673.5, "completions/mean_terminated_length": 3673.5, "completions/min_length": 2296.0, "completions/min_terminated_length": 2296.0, "epoch": 0.026069701103807515, "grad_norm": 0.0, "learning_rate": 4.7499999999999995e-07, "loss": 0.0, "num_tokens": 5727633.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1051.0, "completions/max_terminated_length": 1051.0, "completions/mean_length": 847.0, "completions/mean_terminated_length": 847.0, "completions/min_length": 643.0, "completions/min_terminated_length": 643.0, "epoch": 0.026094505767084213, "grad_norm": 0.0, "learning_rate": 4.7449999999999997e-07, "loss": 0.0, "num_tokens": 5730127.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1052 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 848.0, "completions/max_terminated_length": 848.0, "completions/mean_length": 659.0, "completions/mean_terminated_length": 659.0, "completions/min_length": 470.0, "completions/min_terminated_length": 470.0, "epoch": 0.026119310430360906, "grad_norm": 0.0, "learning_rate": 4.7399999999999993e-07, "loss": 0.0, "num_tokens": 5732343.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1053 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.026144115093637604, "grad_norm": 0.0, "learning_rate": 4.7349999999999995e-07, "loss": 0.0, "num_tokens": 5733271.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3561.0, "completions/max_terminated_length": 3561.0, "completions/mean_length": 3017.5, "completions/mean_terminated_length": 3017.5, "completions/min_length": 2474.0, "completions/min_terminated_length": 2474.0, "epoch": 0.0261689197569143, "grad_norm": 0.0, "learning_rate": 4.7299999999999996e-07, "loss": 0.0, "num_tokens": 5740176.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4147.0, "completions/mean_length": 6169.5, "completions/mean_terminated_length": 4147.0, "completions/min_length": 4147.0, "completions/min_terminated_length": 4147.0, "epoch": 0.026193724420190995, "grad_norm": 0.0, "learning_rate": 4.725e-07, "loss": 0.0, "num_tokens": 5745163.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1950.0, "completions/mean_length": 5071.0, "completions/mean_terminated_length": 1950.0, "completions/min_length": 1950.0, "completions/min_terminated_length": 1950.0, "epoch": 0.026218529083467692, "grad_norm": 6.975522518157959, "learning_rate": 4.7199999999999994e-07, "loss": -0.707, "num_tokens": 5748071.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2328.0, "completions/mean_length": 5260.0, "completions/mean_terminated_length": 2328.0, "completions/min_length": 2328.0, "completions/min_terminated_length": 2328.0, "epoch": 0.02624333374674439, "grad_norm": 0.0, "learning_rate": 4.7149999999999995e-07, "loss": 0.0, "num_tokens": 5751447.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1058 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.026268138410021083, "grad_norm": 0.0, "learning_rate": 4.7099999999999997e-07, "loss": 0.0, "num_tokens": 5752415.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7591.0, "completions/max_terminated_length": 7591.0, "completions/mean_length": 7078.5, "completions/mean_terminated_length": 7078.5, "completions/min_length": 6566.0, "completions/min_terminated_length": 6566.0, "epoch": 0.02629294307329778, "grad_norm": 0.0, "learning_rate": 4.7049999999999993e-07, "loss": 0.0, "num_tokens": 5767524.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1060 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4388.0, "completions/mean_length": 6290.0, "completions/mean_terminated_length": 4388.0, "completions/min_length": 4388.0, "completions/min_terminated_length": 4388.0, "epoch": 0.026317747736574477, "grad_norm": 0.0, "learning_rate": 4.6999999999999995e-07, "loss": 0.0, "num_tokens": 5772790.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7845.0, "completions/max_terminated_length": 7845.0, "completions/mean_length": 6030.0, "completions/mean_terminated_length": 6030.0, "completions/min_length": 4215.0, "completions/min_terminated_length": 4215.0, "epoch": 0.02634255239985117, "grad_norm": 0.0, "learning_rate": 4.6949999999999996e-07, "loss": 0.0, "num_tokens": 5785804.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4267.0, "completions/max_terminated_length": 4267.0, "completions/mean_length": 3956.5, "completions/mean_terminated_length": 3956.5, "completions/min_length": 3646.0, "completions/min_terminated_length": 3646.0, "epoch": 0.02636735706312787, "grad_norm": 0.0, "learning_rate": 4.689999999999999e-07, "loss": 0.0, "num_tokens": 5794595.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1063 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.026392161726404566, "grad_norm": 0.0, "learning_rate": 4.685e-07, "loss": 0.0, "num_tokens": 5795415.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 631.0, "completions/max_terminated_length": 631.0, "completions/mean_length": 618.0, "completions/mean_terminated_length": 618.0, "completions/min_length": 605.0, "completions/min_terminated_length": 605.0, "epoch": 0.02641696638968126, "grad_norm": 0.0, "learning_rate": 4.68e-07, "loss": 0.0, "num_tokens": 5797467.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6141.0, "completions/max_terminated_length": 6141.0, "completions/mean_length": 6139.5, "completions/mean_terminated_length": 6139.5, "completions/min_length": 6138.0, "completions/min_terminated_length": 6138.0, "epoch": 0.026441771052957957, "grad_norm": 0.0, "learning_rate": 4.675e-07, "loss": 0.0, "num_tokens": 5810828.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1066 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02646657571623465, "grad_norm": 0.0, "learning_rate": 4.67e-07, "loss": 0.0, "num_tokens": 5811742.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6335.0, "completions/mean_length": 7263.5, "completions/mean_terminated_length": 6335.0, "completions/min_length": 6335.0, "completions/min_terminated_length": 6335.0, "epoch": 0.026491380379511348, "grad_norm": 3.5383622646331787, "learning_rate": 4.665e-07, "loss": -0.707, "num_tokens": 5819215.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1068 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.026516185042788045, "grad_norm": 0.0, "learning_rate": 4.66e-07, "loss": 0.0, "num_tokens": 5820185.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1069 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5176.0, "completions/mean_length": 6684.0, "completions/mean_terminated_length": 5176.0, "completions/min_length": 5176.0, "completions/min_terminated_length": 5176.0, "epoch": 0.02654098970606474, "grad_norm": 3.6374614238739014, "learning_rate": 4.655e-07, "loss": -0.707, "num_tokens": 5826285.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1070 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7305.0, "completions/max_terminated_length": 7305.0, "completions/mean_length": 6393.5, "completions/mean_terminated_length": 6393.5, "completions/min_length": 5482.0, "completions/min_terminated_length": 5482.0, "epoch": 0.026565794369341436, "grad_norm": 0.0, "learning_rate": 4.65e-07, "loss": 0.0, "num_tokens": 5839918.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 597.0, "completions/max_terminated_length": 597.0, "completions/mean_length": 530.5, "completions/mean_terminated_length": 530.5, "completions/min_length": 464.0, "completions/min_terminated_length": 464.0, "epoch": 0.026590599032618133, "grad_norm": 0.0, "learning_rate": 4.645e-07, "loss": 0.0, "num_tokens": 5841805.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4207.0, "completions/max_terminated_length": 4207.0, "completions/mean_length": 2737.5, "completions/mean_terminated_length": 2737.5, "completions/min_length": 1268.0, "completions/min_terminated_length": 1268.0, "epoch": 0.026615403695894827, "grad_norm": 0.0, "learning_rate": 4.64e-07, "loss": 0.0, "num_tokens": 5848122.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1073 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.026640208359171524, "grad_norm": 0.0, "learning_rate": 4.635e-07, "loss": 0.0, "num_tokens": 5849106.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5639.0, "completions/max_terminated_length": 5639.0, "completions/mean_length": 3440.0, "completions/mean_terminated_length": 3440.0, "completions/min_length": 1241.0, "completions/min_terminated_length": 1241.0, "epoch": 0.02666501302244822, "grad_norm": 0.0, "learning_rate": 4.63e-07, "loss": 0.0, "num_tokens": 5856888.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4389.0, "completions/mean_length": 6290.5, "completions/mean_terminated_length": 4389.0, "completions/min_length": 4389.0, "completions/min_terminated_length": 4389.0, "epoch": 0.026689817685724915, "grad_norm": 4.2470903396606445, "learning_rate": 4.625e-07, "loss": -0.707, "num_tokens": 5862577.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4564.0, "completions/max_terminated_length": 4564.0, "completions/mean_length": 3792.0, "completions/mean_terminated_length": 3792.0, "completions/min_length": 3020.0, "completions/min_terminated_length": 3020.0, "epoch": 0.026714622349001613, "grad_norm": 0.0, "learning_rate": 4.62e-07, "loss": 0.0, "num_tokens": 5871079.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1077 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6654.0, "completions/mean_length": 7423.0, "completions/mean_terminated_length": 6654.0, "completions/min_length": 6654.0, "completions/min_terminated_length": 6654.0, "epoch": 0.02673942701227831, "grad_norm": 3.1731505393981934, "learning_rate": 4.615e-07, "loss": -0.707, "num_tokens": 5878573.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6325.0, "completions/mean_length": 7258.5, "completions/mean_terminated_length": 6325.0, "completions/min_length": 6325.0, "completions/min_terminated_length": 6325.0, "epoch": 0.026764231675555004, "grad_norm": 0.0, "learning_rate": 4.61e-07, "loss": 0.0, "num_tokens": 5885732.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1079 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0267890363388317, "grad_norm": 0.0, "learning_rate": 4.605e-07, "loss": 0.0, "num_tokens": 5886688.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1080 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2083.0, "completions/max_terminated_length": 2083.0, "completions/mean_length": 1656.0, "completions/mean_terminated_length": 1656.0, "completions/min_length": 1229.0, "completions/min_terminated_length": 1229.0, "epoch": 0.026813841002108395, "grad_norm": 0.0, "learning_rate": 4.6e-07, "loss": 0.0, "num_tokens": 5890904.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1081 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.026838645665385092, "grad_norm": 0.0, "learning_rate": 4.595e-07, "loss": 0.0, "num_tokens": 5891880.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1082 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6567.0, "completions/mean_length": 7379.5, "completions/mean_terminated_length": 6567.0, "completions/min_length": 6567.0, "completions/min_terminated_length": 6567.0, "epoch": 0.02686345032866179, "grad_norm": 3.683047294616699, "learning_rate": 4.59e-07, "loss": -0.707, "num_tokens": 5899773.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1083 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5811.0, "completions/mean_length": 7001.5, "completions/mean_terminated_length": 5811.0, "completions/min_length": 5811.0, "completions/min_terminated_length": 5811.0, "epoch": 0.026888254991938483, "grad_norm": 3.5721607208251953, "learning_rate": 4.585e-07, "loss": -0.707, "num_tokens": 5906414.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1084 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02691305965521518, "grad_norm": 0.0, "learning_rate": 4.58e-07, "loss": 0.0, "num_tokens": 5907332.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5384.0, "completions/max_terminated_length": 5384.0, "completions/mean_length": 4474.5, "completions/mean_terminated_length": 4474.5, "completions/min_length": 3565.0, "completions/min_terminated_length": 3565.0, "epoch": 0.026937864318491878, "grad_norm": 0.0, "learning_rate": 4.575e-07, "loss": 0.0, "num_tokens": 5917115.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3877.0, "completions/max_terminated_length": 3877.0, "completions/mean_length": 3220.5, "completions/mean_terminated_length": 3220.5, "completions/min_length": 2564.0, "completions/min_terminated_length": 2564.0, "epoch": 0.02696266898176857, "grad_norm": 0.0, "learning_rate": 4.57e-07, "loss": 0.0, "num_tokens": 5924374.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1087 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1631.0, "completions/max_terminated_length": 1631.0, "completions/mean_length": 1490.5, "completions/mean_terminated_length": 1490.5, "completions/min_length": 1350.0, "completions/min_terminated_length": 1350.0, "epoch": 0.02698747364504527, "grad_norm": 0.0, "learning_rate": 4.565e-07, "loss": 0.0, "num_tokens": 5928233.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1088 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2948.0, "completions/max_terminated_length": 2948.0, "completions/mean_length": 2626.5, "completions/mean_terminated_length": 2626.5, "completions/min_length": 2305.0, "completions/min_terminated_length": 2305.0, "epoch": 0.027012278308321966, "grad_norm": 0.0, "learning_rate": 4.56e-07, "loss": 0.0, "num_tokens": 5934440.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1089 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 764.0, "completions/max_terminated_length": 764.0, "completions/mean_length": 762.0, "completions/mean_terminated_length": 762.0, "completions/min_length": 760.0, "completions/min_terminated_length": 760.0, "epoch": 0.02703708297159866, "grad_norm": 0.0, "learning_rate": 4.5549999999999997e-07, "loss": 0.0, "num_tokens": 5936854.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1090 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4601.0, "completions/mean_length": 6396.5, "completions/mean_terminated_length": 4601.0, "completions/min_length": 4601.0, "completions/min_terminated_length": 4601.0, "epoch": 0.027061887634875357, "grad_norm": 3.73543381690979, "learning_rate": 4.55e-07, "loss": -0.707, "num_tokens": 5942319.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7016.0, "completions/mean_length": 7604.0, "completions/mean_terminated_length": 7016.0, "completions/min_length": 7016.0, "completions/min_terminated_length": 7016.0, "epoch": 0.027086692298152054, "grad_norm": 0.0, "learning_rate": 4.545e-07, "loss": 0.0, "num_tokens": 5950393.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1092 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5661.0, "completions/max_terminated_length": 5661.0, "completions/mean_length": 4668.0, "completions/mean_terminated_length": 4668.0, "completions/min_length": 3675.0, "completions/min_terminated_length": 3675.0, "epoch": 0.027111496961428748, "grad_norm": 0.0, "learning_rate": 4.54e-07, "loss": 0.0, "num_tokens": 5960523.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3953.0, "completions/max_terminated_length": 3953.0, "completions/mean_length": 3396.0, "completions/mean_terminated_length": 3396.0, "completions/min_length": 2839.0, "completions/min_terminated_length": 2839.0, "epoch": 0.027136301624705445, "grad_norm": 0.0, "learning_rate": 4.535e-07, "loss": 0.0, "num_tokens": 5968265.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 887.0, "completions/max_terminated_length": 887.0, "completions/mean_length": 824.0, "completions/mean_terminated_length": 824.0, "completions/min_length": 761.0, "completions/min_terminated_length": 761.0, "epoch": 0.02716110628798214, "grad_norm": 0.0, "learning_rate": 4.53e-07, "loss": 0.0, "num_tokens": 5970793.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5052.0, "completions/max_terminated_length": 5052.0, "completions/mean_length": 4474.5, "completions/mean_terminated_length": 4474.5, "completions/min_length": 3897.0, "completions/min_terminated_length": 3897.0, "epoch": 0.027185910951258836, "grad_norm": 2.699639081954956, "learning_rate": 4.525e-07, "loss": 0.0912, "num_tokens": 5980732.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1096 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.027210715614535533, "grad_norm": 0.0, "learning_rate": 4.5199999999999997e-07, "loss": 0.0, "num_tokens": 5981594.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6821.0, "completions/max_terminated_length": 6821.0, "completions/mean_length": 6803.0, "completions/mean_terminated_length": 6803.0, "completions/min_length": 6785.0, "completions/min_terminated_length": 6785.0, "epoch": 0.027235520277812227, "grad_norm": 0.0, "learning_rate": 4.515e-07, "loss": 0.0, "num_tokens": 5996210.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7065.0, "completions/mean_length": 7628.5, "completions/mean_terminated_length": 7065.0, "completions/min_length": 7065.0, "completions/min_terminated_length": 7065.0, "epoch": 0.027260324941088925, "grad_norm": 0.0, "learning_rate": 4.51e-07, "loss": 0.0, "num_tokens": 6004291.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6562.0, "completions/mean_length": 7377.0, "completions/mean_terminated_length": 6562.0, "completions/min_length": 6562.0, "completions/min_terminated_length": 6562.0, "epoch": 0.027285129604365622, "grad_norm": 2.9623069763183594, "learning_rate": 4.505e-07, "loss": -0.707, "num_tokens": 6011775.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7534.0, "completions/mean_length": 7863.0, "completions/mean_terminated_length": 7534.0, "completions/min_length": 7534.0, "completions/min_terminated_length": 7534.0, "epoch": 0.027309934267642316, "grad_norm": 3.2245376110076904, "learning_rate": 4.5e-07, "loss": -0.707, "num_tokens": 6020179.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1101 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.027334738930919013, "grad_norm": 0.0, "learning_rate": 4.495e-07, "loss": 0.0, "num_tokens": 6021055.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1102 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5984.0, "completions/max_terminated_length": 5984.0, "completions/mean_length": 5269.5, "completions/mean_terminated_length": 5269.5, "completions/min_length": 4555.0, "completions/min_terminated_length": 4555.0, "epoch": 0.02735954359419571, "grad_norm": 2.5761337280273438, "learning_rate": 4.49e-07, "loss": 0.0959, "num_tokens": 6032490.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1103 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 226.0, "completions/mean_length": 4209.0, "completions/mean_terminated_length": 226.0, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "epoch": 0.027384348257472404, "grad_norm": 0.0, "learning_rate": 4.4849999999999997e-07, "loss": 0.0, "num_tokens": 6033628.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3464.0, "completions/max_terminated_length": 3464.0, "completions/mean_length": 3304.5, "completions/mean_terminated_length": 3304.5, "completions/min_length": 3145.0, "completions/min_terminated_length": 3145.0, "epoch": 0.0274091529207491, "grad_norm": 0.0, "learning_rate": 4.48e-07, "loss": 0.0, "num_tokens": 6041069.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6507.0, "completions/mean_length": 7349.5, "completions/mean_terminated_length": 6507.0, "completions/min_length": 6507.0, "completions/min_terminated_length": 6507.0, "epoch": 0.0274339575840258, "grad_norm": 0.0, "learning_rate": 4.475e-07, "loss": 0.0, "num_tokens": 6048454.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7873.0, "completions/max_terminated_length": 7873.0, "completions/mean_length": 7096.0, "completions/mean_terminated_length": 7096.0, "completions/min_length": 6319.0, "completions/min_terminated_length": 6319.0, "epoch": 0.027458762247302492, "grad_norm": 2.377234935760498, "learning_rate": 4.4699999999999997e-07, "loss": 0.0774, "num_tokens": 6063604.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1107 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02748356691057919, "grad_norm": 0.0, "learning_rate": 4.465e-07, "loss": 0.0, "num_tokens": 6064666.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2506.0, "completions/max_terminated_length": 2506.0, "completions/mean_length": 1816.5, "completions/mean_terminated_length": 1816.5, "completions/min_length": 1127.0, "completions/min_terminated_length": 1127.0, "epoch": 0.027508371573855887, "grad_norm": 0.0, "learning_rate": 4.46e-07, "loss": 0.0, "num_tokens": 6069119.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7918.0, "completions/max_terminated_length": 7918.0, "completions/mean_length": 7715.5, "completions/mean_terminated_length": 7715.5, "completions/min_length": 7513.0, "completions/min_terminated_length": 7513.0, "epoch": 0.02753317623713258, "grad_norm": 2.527945041656494, "learning_rate": 4.455e-07, "loss": 0.0186, "num_tokens": 6085630.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3790.0, "completions/max_terminated_length": 3790.0, "completions/mean_length": 2473.5, "completions/mean_terminated_length": 2473.5, "completions/min_length": 1157.0, "completions/min_terminated_length": 1157.0, "epoch": 0.027557980900409278, "grad_norm": 0.0, "learning_rate": 4.45e-07, "loss": 0.0, "num_tokens": 6091517.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3566.0, "completions/mean_length": 5879.0, "completions/mean_terminated_length": 3566.0, "completions/min_length": 3566.0, "completions/min_terminated_length": 3566.0, "epoch": 0.02758278556368597, "grad_norm": 0.0, "learning_rate": 4.445e-07, "loss": 0.0, "num_tokens": 6096035.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3193.0, "completions/max_terminated_length": 3193.0, "completions/mean_length": 2245.5, "completions/mean_terminated_length": 2245.5, "completions/min_length": 1298.0, "completions/min_terminated_length": 1298.0, "epoch": 0.02760759022696267, "grad_norm": 0.0, "learning_rate": 4.44e-07, "loss": 0.0, "num_tokens": 6101374.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1113 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7358.0, "completions/max_terminated_length": 7358.0, "completions/mean_length": 5782.0, "completions/mean_terminated_length": 5782.0, "completions/min_length": 4206.0, "completions/min_terminated_length": 4206.0, "epoch": 0.027632394890239366, "grad_norm": 0.0, "learning_rate": 4.4349999999999997e-07, "loss": 0.0, "num_tokens": 6113874.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3657.0, "completions/max_terminated_length": 3657.0, "completions/mean_length": 3409.5, "completions/mean_terminated_length": 3409.5, "completions/min_length": 3162.0, "completions/min_terminated_length": 3162.0, "epoch": 0.02765719955351606, "grad_norm": 2.736707925796509, "learning_rate": 4.43e-07, "loss": 0.0513, "num_tokens": 6121537.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3944.0, "completions/mean_length": 6068.0, "completions/mean_terminated_length": 3944.0, "completions/min_length": 3944.0, "completions/min_terminated_length": 3944.0, "epoch": 0.027682004216792757, "grad_norm": 4.599496841430664, "learning_rate": 4.425e-07, "loss": -0.707, "num_tokens": 6126303.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1116 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.027706808880069454, "grad_norm": 0.0, "learning_rate": 4.4199999999999996e-07, "loss": 0.0, "num_tokens": 6127237.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2244.0, "completions/max_terminated_length": 2244.0, "completions/mean_length": 1708.5, "completions/mean_terminated_length": 1708.5, "completions/min_length": 1173.0, "completions/min_terminated_length": 1173.0, "epoch": 0.027731613543346148, "grad_norm": 0.0, "learning_rate": 4.415e-07, "loss": 0.0, "num_tokens": 6131508.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5393.0, "completions/max_terminated_length": 5393.0, "completions/mean_length": 4319.5, "completions/mean_terminated_length": 4319.5, "completions/min_length": 3246.0, "completions/min_terminated_length": 3246.0, "epoch": 0.027756418206622845, "grad_norm": 0.0, "learning_rate": 4.41e-07, "loss": 0.0, "num_tokens": 6141099.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1119 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.027781222869899543, "grad_norm": 0.0, "learning_rate": 4.405e-07, "loss": 0.0, "num_tokens": 6141995.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1120 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.027806027533176236, "grad_norm": 0.0, "learning_rate": 4.3999999999999997e-07, "loss": 0.0, "num_tokens": 6142873.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2743.0, "completions/max_terminated_length": 2743.0, "completions/mean_length": 2024.5, "completions/mean_terminated_length": 2024.5, "completions/min_length": 1306.0, "completions/min_terminated_length": 1306.0, "epoch": 0.027830832196452934, "grad_norm": 0.0, "learning_rate": 4.395e-07, "loss": 0.0, "num_tokens": 6147820.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1122 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02785563685972963, "grad_norm": 0.0, "learning_rate": 4.39e-07, "loss": 0.0, "num_tokens": 6148680.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3220.0, "completions/max_terminated_length": 3220.0, "completions/mean_length": 2600.5, "completions/mean_terminated_length": 2600.5, "completions/min_length": 1981.0, "completions/min_terminated_length": 1981.0, "epoch": 0.027880441523006325, "grad_norm": 0.0, "learning_rate": 4.3849999999999996e-07, "loss": 0.0, "num_tokens": 6154827.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2279.0, "completions/max_terminated_length": 2279.0, "completions/mean_length": 2271.0, "completions/mean_terminated_length": 2271.0, "completions/min_length": 2263.0, "completions/min_terminated_length": 2263.0, "epoch": 0.027905246186283022, "grad_norm": 0.0, "learning_rate": 4.38e-07, "loss": 0.0, "num_tokens": 6160207.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6967.0, "completions/max_terminated_length": 6967.0, "completions/mean_length": 6460.0, "completions/mean_terminated_length": 6460.0, "completions/min_length": 5953.0, "completions/min_terminated_length": 5953.0, "epoch": 0.027930050849559716, "grad_norm": 2.7781898975372314, "learning_rate": 4.375e-07, "loss": -0.0555, "num_tokens": 6173965.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2353.0, "completions/max_terminated_length": 2353.0, "completions/mean_length": 2059.5, "completions/mean_terminated_length": 2059.5, "completions/min_length": 1766.0, "completions/min_terminated_length": 1766.0, "epoch": 0.027954855512836413, "grad_norm": 0.0, "learning_rate": 4.3699999999999996e-07, "loss": 0.0, "num_tokens": 6178990.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5760.0, "completions/max_terminated_length": 5760.0, "completions/mean_length": 5490.0, "completions/mean_terminated_length": 5490.0, "completions/min_length": 5220.0, "completions/min_terminated_length": 5220.0, "epoch": 0.02797966017611311, "grad_norm": 2.267627716064453, "learning_rate": 4.3649999999999997e-07, "loss": 0.0348, "num_tokens": 6190828.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1636.0, "completions/max_terminated_length": 1636.0, "completions/mean_length": 1480.5, "completions/mean_terminated_length": 1480.5, "completions/min_length": 1325.0, "completions/min_terminated_length": 1325.0, "epoch": 0.028004464839389804, "grad_norm": 0.0, "learning_rate": 4.36e-07, "loss": 0.0, "num_tokens": 6194685.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2194.0, "completions/max_terminated_length": 2194.0, "completions/mean_length": 2118.5, "completions/mean_terminated_length": 2118.5, "completions/min_length": 2043.0, "completions/min_terminated_length": 2043.0, "epoch": 0.0280292695026665, "grad_norm": 0.0, "learning_rate": 4.355e-07, "loss": 0.0, "num_tokens": 6199862.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6159.0, "completions/mean_length": 7175.5, "completions/mean_terminated_length": 6159.0, "completions/min_length": 6159.0, "completions/min_terminated_length": 6159.0, "epoch": 0.0280540741659432, "grad_norm": 0.0, "learning_rate": 4.3499999999999996e-07, "loss": 0.0, "num_tokens": 6206957.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1131 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8133.0, "completions/max_terminated_length": 8133.0, "completions/mean_length": 5032.5, "completions/mean_terminated_length": 5032.5, "completions/min_length": 1932.0, "completions/min_terminated_length": 1932.0, "epoch": 0.028078878829219892, "grad_norm": 0.0, "learning_rate": 4.345e-07, "loss": 0.0, "num_tokens": 6217904.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1132 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02810368349249659, "grad_norm": 0.0, "learning_rate": 4.34e-07, "loss": 0.0, "num_tokens": 6218858.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1484.0, "completions/max_terminated_length": 1484.0, "completions/mean_length": 1314.5, "completions/mean_terminated_length": 1314.5, "completions/min_length": 1145.0, "completions/min_terminated_length": 1145.0, "epoch": 0.028128488155773287, "grad_norm": 4.729451656341553, "learning_rate": 4.3349999999999996e-07, "loss": -0.0912, "num_tokens": 6222361.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1134 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02815329281904998, "grad_norm": 0.0, "learning_rate": 4.3299999999999997e-07, "loss": 0.0, "num_tokens": 6223229.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1135 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028178097482326678, "grad_norm": 0.0, "learning_rate": 4.325e-07, "loss": 0.0, "num_tokens": 6224137.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1136 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028202902145603375, "grad_norm": 0.0, "learning_rate": 4.3199999999999995e-07, "loss": 0.0, "num_tokens": 6225279.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1137 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02822770680888007, "grad_norm": 0.0, "learning_rate": 4.3149999999999997e-07, "loss": 0.0, "num_tokens": 6226271.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2901.0, "completions/max_terminated_length": 2901.0, "completions/mean_length": 2784.5, "completions/mean_terminated_length": 2784.5, "completions/min_length": 2668.0, "completions/min_terminated_length": 2668.0, "epoch": 0.028252511472156766, "grad_norm": 0.0, "learning_rate": 4.31e-07, "loss": 0.0, "num_tokens": 6232706.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5299.0, "completions/mean_length": 6745.5, "completions/mean_terminated_length": 5299.0, "completions/min_length": 5299.0, "completions/min_terminated_length": 5299.0, "epoch": 0.02827731613543346, "grad_norm": 4.6091437339782715, "learning_rate": 4.305e-07, "loss": -0.707, "num_tokens": 6239531.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7383.0, "completions/max_terminated_length": 7383.0, "completions/mean_length": 6643.5, "completions/mean_terminated_length": 6643.5, "completions/min_length": 5904.0, "completions/min_terminated_length": 5904.0, "epoch": 0.028302120798710157, "grad_norm": 0.0, "learning_rate": 4.2999999999999996e-07, "loss": 0.0, "num_tokens": 6253720.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4453.0, "completions/max_terminated_length": 4453.0, "completions/mean_length": 4419.5, "completions/mean_terminated_length": 4419.5, "completions/min_length": 4386.0, "completions/min_terminated_length": 4386.0, "epoch": 0.028326925461986854, "grad_norm": 0.0, "learning_rate": 4.295e-07, "loss": 0.0, "num_tokens": 6263463.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1106.0, "completions/max_terminated_length": 1106.0, "completions/mean_length": 1097.5, "completions/mean_terminated_length": 1097.5, "completions/min_length": 1089.0, "completions/min_terminated_length": 1089.0, "epoch": 0.028351730125263548, "grad_norm": 5.346375465393066, "learning_rate": 4.29e-07, "loss": -0.0055, "num_tokens": 6266492.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1143 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028376534788540245, "grad_norm": 0.0, "learning_rate": 4.2849999999999995e-07, "loss": 0.0, "num_tokens": 6267562.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4916.0, "completions/mean_length": 6554.0, "completions/mean_terminated_length": 4916.0, "completions/min_length": 4916.0, "completions/min_terminated_length": 4916.0, "epoch": 0.028401339451816943, "grad_norm": 3.272869825363159, "learning_rate": 4.2799999999999997e-07, "loss": -0.707, "num_tokens": 6273466.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4042.0, "completions/max_terminated_length": 4042.0, "completions/mean_length": 3368.0, "completions/mean_terminated_length": 3368.0, "completions/min_length": 2694.0, "completions/min_terminated_length": 2694.0, "epoch": 0.028426144115093636, "grad_norm": 0.0, "learning_rate": 4.275e-07, "loss": 0.0, "num_tokens": 6281380.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3583.0, "completions/mean_length": 5887.5, "completions/mean_terminated_length": 3583.0, "completions/min_length": 3583.0, "completions/min_terminated_length": 3583.0, "epoch": 0.028450948778370334, "grad_norm": 4.023141384124756, "learning_rate": 4.2699999999999995e-07, "loss": -0.707, "num_tokens": 6285847.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7022.0, "completions/mean_length": 7607.0, "completions/mean_terminated_length": 7022.0, "completions/min_length": 7022.0, "completions/min_terminated_length": 7022.0, "epoch": 0.02847575344164703, "grad_norm": 0.0, "learning_rate": 4.2649999999999996e-07, "loss": 0.0, "num_tokens": 6293805.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1101.0, "completions/max_terminated_length": 1101.0, "completions/mean_length": 872.0, "completions/mean_terminated_length": 872.0, "completions/min_length": 643.0, "completions/min_terminated_length": 643.0, "epoch": 0.028500558104923725, "grad_norm": 0.0, "learning_rate": 4.26e-07, "loss": 0.0, "num_tokens": 6296397.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1096.0, "completions/max_terminated_length": 1096.0, "completions/mean_length": 896.5, "completions/mean_terminated_length": 896.5, "completions/min_length": 697.0, "completions/min_terminated_length": 697.0, "epoch": 0.028525362768200422, "grad_norm": 0.0, "learning_rate": 4.255e-07, "loss": 0.0, "num_tokens": 6299002.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1150 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02855016743147712, "grad_norm": 0.0, "learning_rate": 4.2499999999999995e-07, "loss": 0.0, "num_tokens": 6299926.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7094.0, "completions/mean_length": 7643.0, "completions/mean_terminated_length": 7094.0, "completions/min_length": 7094.0, "completions/min_terminated_length": 7094.0, "epoch": 0.028574972094753813, "grad_norm": 0.0, "learning_rate": 4.2449999999999997e-07, "loss": 0.0, "num_tokens": 6308152.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1152 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02859977675803051, "grad_norm": 0.0, "learning_rate": 4.24e-07, "loss": 0.0, "num_tokens": 6309162.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2541.0, "completions/max_terminated_length": 2541.0, "completions/mean_length": 2533.5, "completions/mean_terminated_length": 2533.5, "completions/min_length": 2526.0, "completions/min_terminated_length": 2526.0, "epoch": 0.028624581421307204, "grad_norm": 0.0, "learning_rate": 4.2349999999999995e-07, "loss": 0.0, "num_tokens": 6315143.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6272.0, "completions/max_terminated_length": 6272.0, "completions/mean_length": 5865.0, "completions/mean_terminated_length": 5865.0, "completions/min_length": 5458.0, "completions/min_terminated_length": 5458.0, "epoch": 0.0286493860845839, "grad_norm": 0.0, "learning_rate": 4.2299999999999996e-07, "loss": 0.0, "num_tokens": 6327709.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7661.0, "completions/mean_length": 7926.5, "completions/mean_terminated_length": 7661.0, "completions/min_length": 7661.0, "completions/min_terminated_length": 7661.0, "epoch": 0.0286741907478606, "grad_norm": 3.6201109886169434, "learning_rate": 4.225e-07, "loss": -0.707, "num_tokens": 6336238.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1156 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028698995411137292, "grad_norm": 0.0, "learning_rate": 4.2199999999999994e-07, "loss": 0.0, "num_tokens": 6337290.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3213.0, "completions/mean_length": 5702.5, "completions/mean_terminated_length": 3213.0, "completions/min_length": 3213.0, "completions/min_terminated_length": 3213.0, "epoch": 0.02872380007441399, "grad_norm": 0.0, "learning_rate": 4.2149999999999996e-07, "loss": 0.0, "num_tokens": 6341517.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 5104.0, "completions/mean_terminated_length": 2016.0, "completions/min_length": 2016.0, "completions/min_terminated_length": 2016.0, "epoch": 0.028748604737690687, "grad_norm": 5.064286231994629, "learning_rate": 4.2099999999999997e-07, "loss": -0.707, "num_tokens": 6344441.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7136.0, "completions/mean_length": 7664.0, "completions/mean_terminated_length": 7136.0, "completions/min_length": 7136.0, "completions/min_terminated_length": 7136.0, "epoch": 0.02877340940096738, "grad_norm": 0.0, "learning_rate": 4.205e-07, "loss": 0.0, "num_tokens": 6352579.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2149.0, "completions/mean_length": 5170.5, "completions/mean_terminated_length": 2149.0, "completions/min_length": 2149.0, "completions/min_terminated_length": 2149.0, "epoch": 0.028798214064244078, "grad_norm": 0.0, "learning_rate": 4.1999999999999995e-07, "loss": 0.0, "num_tokens": 6355682.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1161 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028823018727520775, "grad_norm": 0.0, "learning_rate": 4.1949999999999996e-07, "loss": 0.0, "num_tokens": 6356536.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1162 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02884782339079747, "grad_norm": 0.0, "learning_rate": 4.19e-07, "loss": 0.0, "num_tokens": 6357460.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1163 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028872628054074166, "grad_norm": 0.0, "learning_rate": 4.1849999999999994e-07, "loss": 0.0, "num_tokens": 6358472.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4769.0, "completions/max_terminated_length": 4769.0, "completions/mean_length": 4297.5, "completions/mean_terminated_length": 4297.5, "completions/min_length": 3826.0, "completions/min_terminated_length": 3826.0, "epoch": 0.028897432717350863, "grad_norm": 0.0, "learning_rate": 4.1799999999999996e-07, "loss": 0.0, "num_tokens": 6367891.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3486.0, "completions/max_terminated_length": 3486.0, "completions/mean_length": 3436.0, "completions/mean_terminated_length": 3436.0, "completions/min_length": 3386.0, "completions/min_terminated_length": 3386.0, "epoch": 0.028922237380627557, "grad_norm": 0.0, "learning_rate": 4.1749999999999997e-07, "loss": 0.0, "num_tokens": 6375609.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1166 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.028947042043904254, "grad_norm": 0.0, "learning_rate": 4.17e-07, "loss": 0.0, "num_tokens": 6376627.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7229.0, "completions/max_terminated_length": 7229.0, "completions/mean_length": 6776.5, "completions/mean_terminated_length": 6776.5, "completions/min_length": 6324.0, "completions/min_terminated_length": 6324.0, "epoch": 0.02897184670718095, "grad_norm": 0.0, "learning_rate": 4.1649999999999995e-07, "loss": 0.0, "num_tokens": 6391318.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2519.0, "completions/max_terminated_length": 2519.0, "completions/mean_length": 1758.5, "completions/mean_terminated_length": 1758.5, "completions/min_length": 998.0, "completions/min_terminated_length": 998.0, "epoch": 0.028996651370457645, "grad_norm": 0.0, "learning_rate": 4.1599999999999997e-07, "loss": 0.0, "num_tokens": 6395689.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7550.0, "completions/max_terminated_length": 7550.0, "completions/mean_length": 5775.0, "completions/mean_terminated_length": 5775.0, "completions/min_length": 4000.0, "completions/min_terminated_length": 4000.0, "epoch": 0.029021456033734343, "grad_norm": 2.8332736492156982, "learning_rate": 4.155e-07, "loss": -0.2173, "num_tokens": 6408205.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1170 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029046260697011037, "grad_norm": 0.0, "learning_rate": 4.1499999999999994e-07, "loss": 0.0, "num_tokens": 6409475.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1171 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029071065360287734, "grad_norm": 0.0, "learning_rate": 4.1449999999999996e-07, "loss": 0.0, "num_tokens": 6410327.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1172 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1942.0, "completions/max_terminated_length": 1942.0, "completions/mean_length": 1796.5, "completions/mean_terminated_length": 1796.5, "completions/min_length": 1651.0, "completions/min_terminated_length": 1651.0, "epoch": 0.02909587002356443, "grad_norm": 0.0, "learning_rate": 4.14e-07, "loss": 0.0, "num_tokens": 6414748.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1173 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029120674686841125, "grad_norm": 0.0, "learning_rate": 4.1349999999999994e-07, "loss": 0.0, "num_tokens": 6415746.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1174 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029145479350117822, "grad_norm": 0.0, "learning_rate": 4.1299999999999995e-07, "loss": 0.0, "num_tokens": 6416664.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6271.0, "completions/max_terminated_length": 6271.0, "completions/mean_length": 4274.0, "completions/mean_terminated_length": 4274.0, "completions/min_length": 2277.0, "completions/min_terminated_length": 2277.0, "epoch": 0.02917028401339452, "grad_norm": 2.6051738262176514, "learning_rate": 4.1249999999999997e-07, "loss": -0.3303, "num_tokens": 6426114.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1176 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029195088676671213, "grad_norm": 0.0, "learning_rate": 4.12e-07, "loss": 0.0, "num_tokens": 6426994.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7453.0, "completions/mean_length": 7822.5, "completions/mean_terminated_length": 7453.0, "completions/min_length": 7453.0, "completions/min_terminated_length": 7453.0, "epoch": 0.02921989333994791, "grad_norm": 0.0, "learning_rate": 4.1149999999999995e-07, "loss": 0.0, "num_tokens": 6435425.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1178 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029244698003224608, "grad_norm": 0.0, "learning_rate": 4.1099999999999996e-07, "loss": 0.0, "num_tokens": 6436345.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3532.0, "completions/max_terminated_length": 3532.0, "completions/mean_length": 2743.5, "completions/mean_terminated_length": 2743.5, "completions/min_length": 1955.0, "completions/min_terminated_length": 1955.0, "epoch": 0.0292695026665013, "grad_norm": 0.0, "learning_rate": 4.105e-07, "loss": 0.0, "num_tokens": 6442734.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2923.0, "completions/mean_length": 5557.5, "completions/mean_terminated_length": 2923.0, "completions/min_length": 2923.0, "completions/min_terminated_length": 2923.0, "epoch": 0.029294307329778, "grad_norm": 0.0, "learning_rate": 4.0999999999999994e-07, "loss": 0.0, "num_tokens": 6446503.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5114.0, "completions/mean_length": 6653.0, "completions/mean_terminated_length": 5114.0, "completions/min_length": 5114.0, "completions/min_terminated_length": 5114.0, "epoch": 0.029319111993054696, "grad_norm": 0.0, "learning_rate": 4.0949999999999995e-07, "loss": 0.0, "num_tokens": 6452671.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 783.0, "completions/max_terminated_length": 783.0, "completions/mean_length": 669.5, "completions/mean_terminated_length": 669.5, "completions/min_length": 556.0, "completions/min_terminated_length": 556.0, "epoch": 0.02934391665633139, "grad_norm": 0.0, "learning_rate": 4.0899999999999997e-07, "loss": 0.0, "num_tokens": 6454830.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1563.0, "completions/max_terminated_length": 1563.0, "completions/mean_length": 1292.5, "completions/mean_terminated_length": 1292.5, "completions/min_length": 1022.0, "completions/min_terminated_length": 1022.0, "epoch": 0.029368721319608087, "grad_norm": 0.0, "learning_rate": 4.0849999999999993e-07, "loss": 0.0, "num_tokens": 6458277.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1299.0, "completions/max_terminated_length": 1299.0, "completions/mean_length": 1149.5, "completions/mean_terminated_length": 1149.5, "completions/min_length": 1000.0, "completions/min_terminated_length": 1000.0, "epoch": 0.02939352598288478, "grad_norm": 0.0, "learning_rate": 4.0799999999999995e-07, "loss": 0.0, "num_tokens": 6461372.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8133.0, "completions/mean_length": 8162.5, "completions/mean_terminated_length": 8133.0, "completions/min_length": 8133.0, "completions/min_terminated_length": 8133.0, "epoch": 0.029418330646161478, "grad_norm": 2.9746053218841553, "learning_rate": 4.0749999999999996e-07, "loss": -0.707, "num_tokens": 6470363.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1857.0, "completions/max_terminated_length": 1857.0, "completions/mean_length": 1581.5, "completions/mean_terminated_length": 1581.5, "completions/min_length": 1306.0, "completions/min_terminated_length": 1306.0, "epoch": 0.029443135309438175, "grad_norm": 4.347383975982666, "learning_rate": 4.07e-07, "loss": 0.1232, "num_tokens": 6474326.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2016.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 1569.5, "completions/mean_terminated_length": 1569.5, "completions/min_length": 1123.0, "completions/min_terminated_length": 1123.0, "epoch": 0.02946793997271487, "grad_norm": 0.0, "learning_rate": 4.0649999999999994e-07, "loss": 0.0, "num_tokens": 6478301.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7904.0, "completions/max_terminated_length": 7904.0, "completions/mean_length": 6354.5, "completions/mean_terminated_length": 6354.5, "completions/min_length": 4805.0, "completions/min_terminated_length": 4805.0, "epoch": 0.029492744635991566, "grad_norm": 0.0, "learning_rate": 4.06e-07, "loss": 0.0, "num_tokens": 6491816.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6261.0, "completions/max_terminated_length": 6261.0, "completions/mean_length": 4200.0, "completions/mean_terminated_length": 4200.0, "completions/min_length": 2139.0, "completions/min_terminated_length": 2139.0, "epoch": 0.029517549299268264, "grad_norm": 0.0, "learning_rate": 4.055e-07, "loss": 0.0, "num_tokens": 6501156.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1190 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029542353962544957, "grad_norm": 0.0, "learning_rate": 4.05e-07, "loss": 0.0, "num_tokens": 6502024.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4517.0, "completions/mean_length": 6354.5, "completions/mean_terminated_length": 4517.0, "completions/min_length": 4517.0, "completions/min_terminated_length": 4517.0, "epoch": 0.029567158625821655, "grad_norm": 0.0, "learning_rate": 4.045e-07, "loss": 0.0, "num_tokens": 6507485.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1753.0, "completions/mean_length": 4972.5, "completions/mean_terminated_length": 1753.0, "completions/min_length": 1753.0, "completions/min_terminated_length": 1753.0, "epoch": 0.029591963289098352, "grad_norm": 5.2660088539123535, "learning_rate": 4.04e-07, "loss": -0.707, "num_tokens": 6510168.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1100.0, "completions/max_terminated_length": 1100.0, "completions/mean_length": 1030.5, "completions/mean_terminated_length": 1030.5, "completions/min_length": 961.0, "completions/min_terminated_length": 961.0, "epoch": 0.029616767952375046, "grad_norm": 0.0, "learning_rate": 4.0350000000000003e-07, "loss": 0.0, "num_tokens": 6513071.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4250.0, "completions/max_terminated_length": 4250.0, "completions/mean_length": 3314.5, "completions/mean_terminated_length": 3314.5, "completions/min_length": 2379.0, "completions/min_terminated_length": 2379.0, "epoch": 0.029641572615651743, "grad_norm": 0.0, "learning_rate": 4.03e-07, "loss": 0.0, "num_tokens": 6520634.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1195 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.02966637727892844, "grad_norm": 0.0, "learning_rate": 4.025e-07, "loss": 0.0, "num_tokens": 6521598.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7542.0, "completions/mean_length": 7867.0, "completions/mean_terminated_length": 7542.0, "completions/min_length": 7542.0, "completions/min_terminated_length": 7542.0, "epoch": 0.029691181942205134, "grad_norm": 0.0, "learning_rate": 4.02e-07, "loss": 0.0, "num_tokens": 6530030.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5713.0, "completions/max_terminated_length": 5713.0, "completions/mean_length": 5506.5, "completions/mean_terminated_length": 5506.5, "completions/min_length": 5300.0, "completions/min_terminated_length": 5300.0, "epoch": 0.02971598660548183, "grad_norm": 3.010468006134033, "learning_rate": 4.015e-07, "loss": 0.0265, "num_tokens": 6541881.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6299.0, "completions/mean_length": 7245.5, "completions/mean_terminated_length": 6299.0, "completions/min_length": 6299.0, "completions/min_terminated_length": 6299.0, "epoch": 0.029740791268758525, "grad_norm": 0.0, "learning_rate": 4.01e-07, "loss": 0.0, "num_tokens": 6549128.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2456.0, "completions/max_terminated_length": 2456.0, "completions/mean_length": 2167.0, "completions/mean_terminated_length": 2167.0, "completions/min_length": 1878.0, "completions/min_terminated_length": 1878.0, "epoch": 0.029765595932035222, "grad_norm": 0.0, "learning_rate": 4.005e-07, "loss": 0.0, "num_tokens": 6554316.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6054.0, "completions/mean_length": 7123.0, "completions/mean_terminated_length": 6054.0, "completions/min_length": 6054.0, "completions/min_terminated_length": 6054.0, "epoch": 0.02979040059531192, "grad_norm": 0.0, "learning_rate": 4e-07, "loss": 0.0, "num_tokens": 6561340.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4983.0, "completions/max_terminated_length": 4983.0, "completions/mean_length": 3464.5, "completions/mean_terminated_length": 3464.5, "completions/min_length": 1946.0, "completions/min_terminated_length": 1946.0, "epoch": 0.029815205258588613, "grad_norm": 0.0, "learning_rate": 3.995e-07, "loss": 0.0, "num_tokens": 6569141.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1202 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5741.0, "completions/mean_length": 6966.5, "completions/mean_terminated_length": 5741.0, "completions/min_length": 5741.0, "completions/min_terminated_length": 5741.0, "epoch": 0.02984000992186531, "grad_norm": 4.395147323608398, "learning_rate": 3.99e-07, "loss": -0.707, "num_tokens": 6575874.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4555.0, "completions/mean_length": 6373.5, "completions/mean_terminated_length": 4555.0, "completions/min_length": 4555.0, "completions/min_terminated_length": 4555.0, "epoch": 0.029864814585142008, "grad_norm": 3.5247750282287598, "learning_rate": 3.9850000000000003e-07, "loss": -0.707, "num_tokens": 6581431.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8096.0, "completions/mean_length": 8144.0, "completions/mean_terminated_length": 8096.0, "completions/min_length": 8096.0, "completions/min_terminated_length": 8096.0, "epoch": 0.0298896192484187, "grad_norm": 3.469520092010498, "learning_rate": 3.98e-07, "loss": -0.707, "num_tokens": 6590373.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6467.0, "completions/max_terminated_length": 6467.0, "completions/mean_length": 5040.0, "completions/mean_terminated_length": 5040.0, "completions/min_length": 3613.0, "completions/min_terminated_length": 3613.0, "epoch": 0.0299144239116954, "grad_norm": 0.0, "learning_rate": 3.975e-07, "loss": 0.0, "num_tokens": 6601321.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1206 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.029939228574972096, "grad_norm": 0.0, "learning_rate": 3.97e-07, "loss": 0.0, "num_tokens": 6602347.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4133.0, "completions/max_terminated_length": 4133.0, "completions/mean_length": 3371.5, "completions/mean_terminated_length": 3371.5, "completions/min_length": 2610.0, "completions/min_terminated_length": 2610.0, "epoch": 0.02996403323824879, "grad_norm": 0.0, "learning_rate": 3.965e-07, "loss": 0.0, "num_tokens": 6609920.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 911.0, "completions/max_terminated_length": 911.0, "completions/mean_length": 739.5, "completions/mean_terminated_length": 739.5, "completions/min_length": 568.0, "completions/min_terminated_length": 568.0, "epoch": 0.029988837901525487, "grad_norm": 0.0, "learning_rate": 3.96e-07, "loss": 0.0, "num_tokens": 6612211.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7708.0, "completions/mean_length": 7950.0, "completions/mean_terminated_length": 7708.0, "completions/min_length": 7708.0, "completions/min_terminated_length": 7708.0, "epoch": 0.030013642564802184, "grad_norm": 4.110317230224609, "learning_rate": 3.955e-07, "loss": -0.707, "num_tokens": 6620847.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1210 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.030038447228078878, "grad_norm": 0.0, "learning_rate": 3.95e-07, "loss": 0.0, "num_tokens": 6621769.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1211 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.030063251891355575, "grad_norm": 0.0, "learning_rate": 3.945e-07, "loss": 0.0, "num_tokens": 6622729.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2668.0, "completions/max_terminated_length": 2668.0, "completions/mean_length": 2419.5, "completions/mean_terminated_length": 2419.5, "completions/min_length": 2171.0, "completions/min_terminated_length": 2171.0, "epoch": 0.030088056554632273, "grad_norm": 0.0, "learning_rate": 3.94e-07, "loss": 0.0, "num_tokens": 6628468.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2572.0, "completions/max_terminated_length": 2572.0, "completions/mean_length": 2484.0, "completions/mean_terminated_length": 2484.0, "completions/min_length": 2396.0, "completions/min_terminated_length": 2396.0, "epoch": 0.030112861217908966, "grad_norm": 0.0, "learning_rate": 3.935e-07, "loss": 0.0, "num_tokens": 6634294.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6459.0, "completions/mean_length": 7325.5, "completions/mean_terminated_length": 6459.0, "completions/min_length": 6459.0, "completions/min_terminated_length": 6459.0, "epoch": 0.030137665881185664, "grad_norm": 3.551469326019287, "learning_rate": 3.93e-07, "loss": -0.707, "num_tokens": 6641621.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1215 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.030162470544462357, "grad_norm": 0.0, "learning_rate": 3.925e-07, "loss": 0.0, "num_tokens": 6642517.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3678.0, "completions/max_terminated_length": 3678.0, "completions/mean_length": 3157.0, "completions/mean_terminated_length": 3157.0, "completions/min_length": 2636.0, "completions/min_terminated_length": 2636.0, "epoch": 0.030187275207739055, "grad_norm": 0.0, "learning_rate": 3.92e-07, "loss": 0.0, "num_tokens": 6649879.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1217 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3928.0, "completions/max_terminated_length": 3928.0, "completions/mean_length": 2676.5, "completions/mean_terminated_length": 2676.5, "completions/min_length": 1425.0, "completions/min_terminated_length": 1425.0, "epoch": 0.030212079871015752, "grad_norm": 0.0, "learning_rate": 3.915e-07, "loss": 0.0, "num_tokens": 6656108.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5357.0, "completions/max_terminated_length": 5357.0, "completions/mean_length": 3868.5, "completions/mean_terminated_length": 3868.5, "completions/min_length": 2380.0, "completions/min_terminated_length": 2380.0, "epoch": 0.030236884534292446, "grad_norm": 0.0, "learning_rate": 3.91e-07, "loss": 0.0, "num_tokens": 6664771.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 989.0, "completions/max_terminated_length": 989.0, "completions/mean_length": 832.5, "completions/mean_terminated_length": 832.5, "completions/min_length": 676.0, "completions/min_terminated_length": 676.0, "epoch": 0.030261689197569143, "grad_norm": 0.0, "learning_rate": 3.905e-07, "loss": 0.0, "num_tokens": 6667324.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4416.0, "completions/max_terminated_length": 4416.0, "completions/mean_length": 3464.0, "completions/mean_terminated_length": 3464.0, "completions/min_length": 2512.0, "completions/min_terminated_length": 2512.0, "epoch": 0.03028649386084584, "grad_norm": 0.0, "learning_rate": 3.8999999999999997e-07, "loss": 0.0, "num_tokens": 6675122.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2854.0, "completions/max_terminated_length": 2854.0, "completions/mean_length": 2283.0, "completions/mean_terminated_length": 2283.0, "completions/min_length": 1712.0, "completions/min_terminated_length": 1712.0, "epoch": 0.030311298524122534, "grad_norm": 0.0, "learning_rate": 3.895e-07, "loss": 0.0, "num_tokens": 6680508.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6445.0, "completions/max_terminated_length": 6445.0, "completions/mean_length": 6402.5, "completions/mean_terminated_length": 6402.5, "completions/min_length": 6360.0, "completions/min_terminated_length": 6360.0, "epoch": 0.03033610318739923, "grad_norm": 2.6533493995666504, "learning_rate": 3.89e-07, "loss": -0.0047, "num_tokens": 6694251.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7677.0, "completions/max_terminated_length": 7677.0, "completions/mean_length": 5552.5, "completions/mean_terminated_length": 5552.5, "completions/min_length": 3428.0, "completions/min_terminated_length": 3428.0, "epoch": 0.03036090785067593, "grad_norm": 0.0, "learning_rate": 3.885e-07, "loss": 0.0, "num_tokens": 6706288.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1224 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.030385712513952622, "grad_norm": 0.0, "learning_rate": 3.88e-07, "loss": 0.0, "num_tokens": 6707228.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2889.0, "completions/max_terminated_length": 2889.0, "completions/mean_length": 2632.0, "completions/mean_terminated_length": 2632.0, "completions/min_length": 2375.0, "completions/min_terminated_length": 2375.0, "epoch": 0.03041051717722932, "grad_norm": 3.5186283588409424, "learning_rate": 3.875e-07, "loss": 0.069, "num_tokens": 6713320.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4453.0, "completions/max_terminated_length": 4453.0, "completions/mean_length": 3468.0, "completions/mean_terminated_length": 3468.0, "completions/min_length": 2483.0, "completions/min_terminated_length": 2483.0, "epoch": 0.030435321840506017, "grad_norm": 0.0, "learning_rate": 3.87e-07, "loss": 0.0, "num_tokens": 6721084.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5378.0, "completions/mean_length": 6785.0, "completions/mean_terminated_length": 5378.0, "completions/min_length": 5378.0, "completions/min_terminated_length": 5378.0, "epoch": 0.03046012650378271, "grad_norm": 0.0, "learning_rate": 3.8649999999999997e-07, "loss": 0.0, "num_tokens": 6727354.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5724.0, "completions/max_terminated_length": 5724.0, "completions/mean_length": 5026.0, "completions/mean_terminated_length": 5026.0, "completions/min_length": 4328.0, "completions/min_terminated_length": 4328.0, "epoch": 0.030484931167059408, "grad_norm": 0.0, "learning_rate": 3.86e-07, "loss": 0.0, "num_tokens": 6738246.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1229 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 706.0, "completions/max_terminated_length": 706.0, "completions/mean_length": 612.0, "completions/mean_terminated_length": 612.0, "completions/min_length": 518.0, "completions/min_terminated_length": 518.0, "epoch": 0.0305097358303361, "grad_norm": 0.0, "learning_rate": 3.855e-07, "loss": 0.0, "num_tokens": 6740264.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1230 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0305345404936128, "grad_norm": 0.0, "learning_rate": 3.8499999999999997e-07, "loss": 0.0, "num_tokens": 6741244.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7067.0, "completions/mean_length": 7629.5, "completions/mean_terminated_length": 7067.0, "completions/min_length": 7067.0, "completions/min_terminated_length": 7067.0, "epoch": 0.030559345156889496, "grad_norm": 3.768617868423462, "learning_rate": 3.845e-07, "loss": -0.707, "num_tokens": 6749285.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1232 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03058414982016619, "grad_norm": 0.0, "learning_rate": 3.84e-07, "loss": 0.0, "num_tokens": 6750269.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5067.0, "completions/mean_length": 6629.5, "completions/mean_terminated_length": 5067.0, "completions/min_length": 5067.0, "completions/min_terminated_length": 5067.0, "epoch": 0.030608954483442887, "grad_norm": 3.723696231842041, "learning_rate": 3.835e-07, "loss": -0.707, "num_tokens": 6756252.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1234 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.030633759146719584, "grad_norm": 0.0, "learning_rate": 3.83e-07, "loss": 0.0, "num_tokens": 6757222.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1335.0, "completions/max_terminated_length": 1335.0, "completions/mean_length": 1304.5, "completions/mean_terminated_length": 1304.5, "completions/min_length": 1274.0, "completions/min_terminated_length": 1274.0, "epoch": 0.030658563809996278, "grad_norm": 0.0, "learning_rate": 3.825e-07, "loss": 0.0, "num_tokens": 6760875.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7501.0, "completions/mean_length": 7846.5, "completions/mean_terminated_length": 7501.0, "completions/min_length": 7501.0, "completions/min_terminated_length": 7501.0, "epoch": 0.030683368473272975, "grad_norm": 0.0, "learning_rate": 3.82e-07, "loss": 0.0, "num_tokens": 6769392.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7207.0, "completions/mean_length": 7699.5, "completions/mean_terminated_length": 7207.0, "completions/min_length": 7207.0, "completions/min_terminated_length": 7207.0, "epoch": 0.030708173136549673, "grad_norm": 0.0, "learning_rate": 3.8149999999999997e-07, "loss": 0.0, "num_tokens": 6777423.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4645.0, "completions/max_terminated_length": 4645.0, "completions/mean_length": 4483.0, "completions/mean_terminated_length": 4483.0, "completions/min_length": 4321.0, "completions/min_terminated_length": 4321.0, "epoch": 0.030732977799826366, "grad_norm": 0.0, "learning_rate": 3.81e-07, "loss": 0.0, "num_tokens": 6787443.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1239 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3166.0, "completions/mean_length": 5679.0, "completions/mean_terminated_length": 3166.0, "completions/min_length": 3166.0, "completions/min_terminated_length": 3166.0, "epoch": 0.030757782463103064, "grad_norm": 4.29971981048584, "learning_rate": 3.805e-07, "loss": -0.707, "num_tokens": 6791537.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1240 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03078258712637976, "grad_norm": 0.0, "learning_rate": 3.7999999999999996e-07, "loss": 0.0, "num_tokens": 6792541.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7779.0, "completions/mean_length": 7985.5, "completions/mean_terminated_length": 7779.0, "completions/min_length": 7779.0, "completions/min_terminated_length": 7779.0, "epoch": 0.030807391789656455, "grad_norm": 0.0, "learning_rate": 3.795e-07, "loss": 0.0, "num_tokens": 6802018.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1839.0, "completions/max_terminated_length": 1839.0, "completions/mean_length": 1379.0, "completions/mean_terminated_length": 1379.0, "completions/min_length": 919.0, "completions/min_terminated_length": 919.0, "epoch": 0.030832196452933152, "grad_norm": 0.0, "learning_rate": 3.79e-07, "loss": 0.0, "num_tokens": 6805650.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1243 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.030857001116209846, "grad_norm": 0.0, "learning_rate": 3.785e-07, "loss": 0.0, "num_tokens": 6806612.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 325.0, "completions/mean_terminated_length": 325.0, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "epoch": 0.030881805779486543, "grad_norm": 11.865052223205566, "learning_rate": 3.7799999999999997e-07, "loss": -0.2306, "num_tokens": 6808090.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2226.0, "completions/max_terminated_length": 2226.0, "completions/mean_length": 1752.0, "completions/mean_terminated_length": 1752.0, "completions/min_length": 1278.0, "completions/min_terminated_length": 1278.0, "epoch": 0.03090661044276324, "grad_norm": 0.0, "learning_rate": 3.775e-07, "loss": 0.0, "num_tokens": 6812496.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2136.0, "completions/mean_length": 5164.0, "completions/mean_terminated_length": 2136.0, "completions/min_length": 2136.0, "completions/min_terminated_length": 2136.0, "epoch": 0.030931415106039934, "grad_norm": 5.392582416534424, "learning_rate": 3.77e-07, "loss": -0.707, "num_tokens": 6815440.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6146.0, "completions/max_terminated_length": 6146.0, "completions/mean_length": 5704.5, "completions/mean_terminated_length": 5704.5, "completions/min_length": 5263.0, "completions/min_terminated_length": 5263.0, "epoch": 0.03095621976931663, "grad_norm": 0.0, "learning_rate": 3.7649999999999996e-07, "loss": 0.0, "num_tokens": 6827819.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1248 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03098102443259333, "grad_norm": 0.0, "learning_rate": 3.76e-07, "loss": 0.0, "num_tokens": 6828799.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1249 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031005829095870022, "grad_norm": 0.0, "learning_rate": 3.755e-07, "loss": 0.0, "num_tokens": 6829897.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1978.0, "completions/mean_length": 5085.0, "completions/mean_terminated_length": 1978.0, "completions/min_length": 1978.0, "completions/min_terminated_length": 1978.0, "epoch": 0.03103063375914672, "grad_norm": 6.17829704284668, "learning_rate": 3.75e-07, "loss": -0.707, "num_tokens": 6832821.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1251 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031055438422423417, "grad_norm": 0.0, "learning_rate": 3.7449999999999997e-07, "loss": 0.0, "num_tokens": 6833845.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1252 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3267.0, "completions/mean_length": 5729.5, "completions/mean_terminated_length": 3267.0, "completions/min_length": 3267.0, "completions/min_terminated_length": 3267.0, "epoch": 0.03108024308570011, "grad_norm": 4.898797035217285, "learning_rate": 3.74e-07, "loss": -0.707, "num_tokens": 6837988.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1253 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8156.0, "completions/max_terminated_length": 8156.0, "completions/mean_length": 4815.5, "completions/mean_terminated_length": 4815.5, "completions/min_length": 1475.0, "completions/min_terminated_length": 1475.0, "epoch": 0.031105047748976808, "grad_norm": 0.0, "learning_rate": 3.735e-07, "loss": 0.0, "num_tokens": 6848709.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2013.0, "completions/max_terminated_length": 2013.0, "completions/mean_length": 1899.5, "completions/mean_terminated_length": 1899.5, "completions/min_length": 1786.0, "completions/min_terminated_length": 1786.0, "epoch": 0.031129852412253505, "grad_norm": 0.0, "learning_rate": 3.7299999999999997e-07, "loss": 0.0, "num_tokens": 6853370.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7789.0, "completions/mean_length": 7990.5, "completions/mean_terminated_length": 7789.0, "completions/min_length": 7789.0, "completions/min_terminated_length": 7789.0, "epoch": 0.0311546570755302, "grad_norm": 3.5519778728485107, "learning_rate": 3.725e-07, "loss": -0.707, "num_tokens": 6861983.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1256 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031179461738806896, "grad_norm": 0.0, "learning_rate": 3.72e-07, "loss": 0.0, "num_tokens": 6862955.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 551.0, "completions/max_terminated_length": 551.0, "completions/mean_length": 501.5, "completions/mean_terminated_length": 501.5, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "epoch": 0.03120426640208359, "grad_norm": 0.0, "learning_rate": 3.7149999999999996e-07, "loss": 0.0, "num_tokens": 6864798.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1258 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031229071065360287, "grad_norm": 0.0, "learning_rate": 3.71e-07, "loss": 0.0, "num_tokens": 6865990.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6987.0, "completions/mean_length": 7589.5, "completions/mean_terminated_length": 6987.0, "completions/min_length": 6987.0, "completions/min_terminated_length": 6987.0, "epoch": 0.031253875728636984, "grad_norm": 3.217369556427002, "learning_rate": 3.705e-07, "loss": -0.707, "num_tokens": 6873919.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1260 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3795.0, "completions/max_terminated_length": 3795.0, "completions/mean_length": 2598.0, "completions/mean_terminated_length": 2598.0, "completions/min_length": 1401.0, "completions/min_terminated_length": 1401.0, "epoch": 0.03127868039191368, "grad_norm": 0.0, "learning_rate": 3.7e-07, "loss": 0.0, "num_tokens": 6880291.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1261 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 602.0, "completions/max_terminated_length": 602.0, "completions/mean_length": 543.5, "completions/mean_terminated_length": 543.5, "completions/min_length": 485.0, "completions/min_terminated_length": 485.0, "epoch": 0.03130348505519038, "grad_norm": 0.0, "learning_rate": 3.6949999999999997e-07, "loss": 0.0, "num_tokens": 6882282.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1576.0, "completions/max_terminated_length": 1576.0, "completions/mean_length": 1566.0, "completions/mean_terminated_length": 1566.0, "completions/min_length": 1556.0, "completions/min_terminated_length": 1556.0, "epoch": 0.03132828971846707, "grad_norm": 0.0, "learning_rate": 3.69e-07, "loss": 0.0, "num_tokens": 6886250.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1263 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03135309438174377, "grad_norm": 0.0, "learning_rate": 3.685e-07, "loss": 0.0, "num_tokens": 6887500.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6514.0, "completions/mean_length": 7353.0, "completions/mean_terminated_length": 6514.0, "completions/min_length": 6514.0, "completions/min_terminated_length": 6514.0, "epoch": 0.03137789904502047, "grad_norm": 0.0, "learning_rate": 3.6799999999999996e-07, "loss": 0.0, "num_tokens": 6894984.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1265 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03140270370829716, "grad_norm": 0.0, "learning_rate": 3.675e-07, "loss": 0.0, "num_tokens": 6895904.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1266 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031427508371573855, "grad_norm": 0.0, "learning_rate": 3.67e-07, "loss": 0.0, "num_tokens": 6896884.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1267 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3581.0, "completions/max_terminated_length": 3581.0, "completions/mean_length": 3076.5, "completions/mean_terminated_length": 3076.5, "completions/min_length": 2572.0, "completions/min_terminated_length": 2572.0, "epoch": 0.03145231303485055, "grad_norm": 0.0, "learning_rate": 3.6649999999999995e-07, "loss": 0.0, "num_tokens": 6903855.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1268 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6757.0, "completions/max_terminated_length": 6757.0, "completions/mean_length": 5417.5, "completions/mean_terminated_length": 5417.5, "completions/min_length": 4078.0, "completions/min_terminated_length": 4078.0, "epoch": 0.03147711769812725, "grad_norm": 0.0, "learning_rate": 3.6599999999999997e-07, "loss": 0.0, "num_tokens": 6915514.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1269 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2069.0, "completions/max_terminated_length": 2069.0, "completions/mean_length": 1427.5, "completions/mean_terminated_length": 1427.5, "completions/min_length": 786.0, "completions/min_terminated_length": 786.0, "epoch": 0.03150192236140394, "grad_norm": 0.0, "learning_rate": 3.655e-07, "loss": 0.0, "num_tokens": 6919253.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1270 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2938.0, "completions/max_terminated_length": 2938.0, "completions/mean_length": 2528.0, "completions/mean_terminated_length": 2528.0, "completions/min_length": 2118.0, "completions/min_terminated_length": 2118.0, "epoch": 0.03152672702468064, "grad_norm": 0.0, "learning_rate": 3.65e-07, "loss": 0.0, "num_tokens": 6925251.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1271 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 934.0, "completions/max_terminated_length": 934.0, "completions/mean_length": 758.0, "completions/mean_terminated_length": 758.0, "completions/min_length": 582.0, "completions/min_terminated_length": 582.0, "epoch": 0.03155153168795734, "grad_norm": 0.0, "learning_rate": 3.6449999999999996e-07, "loss": 0.0, "num_tokens": 6927611.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1272 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03157633635123403, "grad_norm": 0.0, "learning_rate": 3.64e-07, "loss": 0.0, "num_tokens": 6928613.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2839.0, "completions/max_terminated_length": 2839.0, "completions/mean_length": 2552.0, "completions/mean_terminated_length": 2552.0, "completions/min_length": 2265.0, "completions/min_terminated_length": 2265.0, "epoch": 0.031601141014510725, "grad_norm": 0.0, "learning_rate": 3.635e-07, "loss": 0.0, "num_tokens": 6934573.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2015.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1850.5, "completions/mean_terminated_length": 1850.5, "completions/min_length": 1686.0, "completions/min_terminated_length": 1686.0, "epoch": 0.031625945677787426, "grad_norm": 0.0, "learning_rate": 3.6299999999999995e-07, "loss": 0.0, "num_tokens": 6939090.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6638.0, "completions/mean_length": 7415.0, "completions/mean_terminated_length": 6638.0, "completions/min_length": 6638.0, "completions/min_terminated_length": 6638.0, "epoch": 0.03165075034106412, "grad_norm": 3.2926483154296875, "learning_rate": 3.6249999999999997e-07, "loss": -0.707, "num_tokens": 6946704.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4233.0, "completions/max_terminated_length": 4233.0, "completions/mean_length": 3283.5, "completions/mean_terminated_length": 3283.5, "completions/min_length": 2334.0, "completions/min_terminated_length": 2334.0, "epoch": 0.031675555004340814, "grad_norm": 3.4410598278045654, "learning_rate": 3.62e-07, "loss": 0.2044, "num_tokens": 6954171.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1277 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031700359667617514, "grad_norm": 0.0, "learning_rate": 3.6149999999999995e-07, "loss": 0.0, "num_tokens": 6955089.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1278 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2839.0, "completions/max_terminated_length": 2839.0, "completions/mean_length": 2079.0, "completions/mean_terminated_length": 2079.0, "completions/min_length": 1319.0, "completions/min_terminated_length": 1319.0, "epoch": 0.03172516433089421, "grad_norm": 0.0, "learning_rate": 3.6099999999999996e-07, "loss": 0.0, "num_tokens": 6960165.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1739.0, "completions/max_terminated_length": 1739.0, "completions/mean_length": 1404.0, "completions/mean_terminated_length": 1404.0, "completions/min_length": 1069.0, "completions/min_terminated_length": 1069.0, "epoch": 0.0317499689941709, "grad_norm": 0.0, "learning_rate": 3.605e-07, "loss": 0.0, "num_tokens": 6963781.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4656.0, "completions/max_terminated_length": 4656.0, "completions/mean_length": 4261.0, "completions/mean_terminated_length": 4261.0, "completions/min_length": 3866.0, "completions/min_terminated_length": 3866.0, "epoch": 0.0317747736574476, "grad_norm": 0.0, "learning_rate": 3.6e-07, "loss": 0.0, "num_tokens": 6973221.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1572.0, "completions/mean_length": 4882.0, "completions/mean_terminated_length": 1572.0, "completions/min_length": 1572.0, "completions/min_terminated_length": 1572.0, "epoch": 0.031799578320724296, "grad_norm": 7.0496439933776855, "learning_rate": 3.5949999999999996e-07, "loss": -0.707, "num_tokens": 6975891.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4657.0, "completions/max_terminated_length": 4657.0, "completions/mean_length": 3934.5, "completions/mean_terminated_length": 3934.5, "completions/min_length": 3212.0, "completions/min_terminated_length": 3212.0, "epoch": 0.03182438298400099, "grad_norm": 0.0, "learning_rate": 3.5899999999999997e-07, "loss": 0.0, "num_tokens": 6984690.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1283 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4836.0, "completions/max_terminated_length": 4836.0, "completions/mean_length": 4757.5, "completions/mean_terminated_length": 4757.5, "completions/min_length": 4679.0, "completions/min_terminated_length": 4679.0, "epoch": 0.03184918764727769, "grad_norm": 0.0, "learning_rate": 3.585e-07, "loss": 0.0, "num_tokens": 6995131.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1284 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.031873992310554385, "grad_norm": 0.0, "learning_rate": 3.5799999999999995e-07, "loss": 0.0, "num_tokens": 6996009.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5357.0, "completions/mean_length": 6774.5, "completions/mean_terminated_length": 5357.0, "completions/min_length": 5357.0, "completions/min_terminated_length": 5357.0, "epoch": 0.03189879697383108, "grad_norm": 0.0, "learning_rate": 3.5749999999999997e-07, "loss": 0.0, "num_tokens": 7002324.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1286 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03192360163710778, "grad_norm": 0.0, "learning_rate": 3.57e-07, "loss": 0.0, "num_tokens": 7003282.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4832.0, "completions/max_terminated_length": 4832.0, "completions/mean_length": 4025.0, "completions/mean_terminated_length": 4025.0, "completions/min_length": 3218.0, "completions/min_terminated_length": 3218.0, "epoch": 0.03194840630038447, "grad_norm": 0.0, "learning_rate": 3.5649999999999994e-07, "loss": 0.0, "num_tokens": 7012320.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2379.0, "completions/max_terminated_length": 2379.0, "completions/mean_length": 1806.0, "completions/mean_terminated_length": 1806.0, "completions/min_length": 1233.0, "completions/min_terminated_length": 1233.0, "epoch": 0.03197321096366117, "grad_norm": 0.0, "learning_rate": 3.5599999999999996e-07, "loss": 0.0, "num_tokens": 7016744.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7666.0, "completions/mean_length": 7929.0, "completions/mean_terminated_length": 7666.0, "completions/min_length": 7666.0, "completions/min_terminated_length": 7666.0, "epoch": 0.03199801562693787, "grad_norm": 3.0547499656677246, "learning_rate": 3.555e-07, "loss": -0.707, "num_tokens": 7025516.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1290 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03202282029021456, "grad_norm": 0.0, "learning_rate": 3.55e-07, "loss": 0.0, "num_tokens": 7026478.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1291 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5904.0, "completions/max_terminated_length": 5904.0, "completions/mean_length": 4492.0, "completions/mean_terminated_length": 4492.0, "completions/min_length": 3080.0, "completions/min_terminated_length": 3080.0, "epoch": 0.032047624953491255, "grad_norm": 0.0, "learning_rate": 3.5449999999999995e-07, "loss": 0.0, "num_tokens": 7036270.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5717.0, "completions/mean_length": 6954.5, "completions/mean_terminated_length": 5717.0, "completions/min_length": 5717.0, "completions/min_terminated_length": 5717.0, "epoch": 0.032072429616767956, "grad_norm": 3.471022844314575, "learning_rate": 3.5399999999999997e-07, "loss": -0.707, "num_tokens": 7042987.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1293 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03209723428004465, "grad_norm": 0.0, "learning_rate": 3.535e-07, "loss": 0.0, "num_tokens": 7043943.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1350.0, "completions/max_terminated_length": 1350.0, "completions/mean_length": 1001.0, "completions/mean_terminated_length": 1001.0, "completions/min_length": 652.0, "completions/min_terminated_length": 652.0, "epoch": 0.03212203894332134, "grad_norm": 0.0, "learning_rate": 3.5299999999999994e-07, "loss": 0.0, "num_tokens": 7046775.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1295 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03214684360659804, "grad_norm": 0.0, "learning_rate": 3.5249999999999996e-07, "loss": 0.0, "num_tokens": 7047745.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1342.0, "completions/max_terminated_length": 1342.0, "completions/mean_length": 1235.5, "completions/mean_terminated_length": 1235.5, "completions/min_length": 1129.0, "completions/min_terminated_length": 1129.0, "epoch": 0.03217164826987474, "grad_norm": 0.0, "learning_rate": 3.52e-07, "loss": 0.0, "num_tokens": 7051080.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1297 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03219645293315143, "grad_norm": 0.0, "learning_rate": 3.5149999999999994e-07, "loss": 0.0, "num_tokens": 7051934.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1298 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2291.0, "completions/max_terminated_length": 2291.0, "completions/mean_length": 2180.5, "completions/mean_terminated_length": 2180.5, "completions/min_length": 2070.0, "completions/min_terminated_length": 2070.0, "epoch": 0.032221257596428125, "grad_norm": 0.0, "learning_rate": 3.5099999999999995e-07, "loss": 0.0, "num_tokens": 7057121.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1984.0, "completions/max_terminated_length": 1984.0, "completions/mean_length": 1295.5, "completions/mean_terminated_length": 1295.5, "completions/min_length": 607.0, "completions/min_terminated_length": 607.0, "epoch": 0.032246062259704826, "grad_norm": 4.8978447914123535, "learning_rate": 3.5049999999999997e-07, "loss": -0.3757, "num_tokens": 7060506.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6126.0, "completions/mean_length": 7159.0, "completions/mean_terminated_length": 6126.0, "completions/min_length": 6126.0, "completions/min_terminated_length": 6126.0, "epoch": 0.03227086692298152, "grad_norm": 3.1253459453582764, "learning_rate": 3.5e-07, "loss": -0.707, "num_tokens": 7067586.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1301 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.032295671586258214, "grad_norm": 0.0, "learning_rate": 3.4949999999999995e-07, "loss": 0.0, "num_tokens": 7068654.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1302 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7700.0, "completions/mean_length": 7946.0, "completions/mean_terminated_length": 7700.0, "completions/min_length": 7700.0, "completions/min_terminated_length": 7700.0, "epoch": 0.032320476249534914, "grad_norm": 0.0, "learning_rate": 3.4899999999999996e-07, "loss": 0.0, "num_tokens": 7077252.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3728.0, "completions/mean_length": 5960.0, "completions/mean_terminated_length": 3728.0, "completions/min_length": 3728.0, "completions/min_terminated_length": 3728.0, "epoch": 0.03234528091281161, "grad_norm": 0.0, "learning_rate": 3.485e-07, "loss": 0.0, "num_tokens": 7081868.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7848.0, "completions/mean_length": 8020.0, "completions/mean_terminated_length": 7848.0, "completions/min_length": 7848.0, "completions/min_terminated_length": 7848.0, "epoch": 0.0323700855760883, "grad_norm": 0.0, "learning_rate": 3.4799999999999994e-07, "loss": 0.0, "num_tokens": 7090606.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2528.0, "completions/max_terminated_length": 2528.0, "completions/mean_length": 1754.5, "completions/mean_terminated_length": 1754.5, "completions/min_length": 981.0, "completions/min_terminated_length": 981.0, "epoch": 0.032394890239365, "grad_norm": 0.0, "learning_rate": 3.4749999999999996e-07, "loss": 0.0, "num_tokens": 7094953.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5961.0, "completions/max_terminated_length": 5961.0, "completions/mean_length": 4131.5, "completions/mean_terminated_length": 4131.5, "completions/min_length": 2302.0, "completions/min_terminated_length": 2302.0, "epoch": 0.032419694902641696, "grad_norm": 0.0, "learning_rate": 3.4699999999999997e-07, "loss": 0.0, "num_tokens": 7104098.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1307 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03244449956591839, "grad_norm": 0.0, "learning_rate": 3.4649999999999993e-07, "loss": 0.0, "num_tokens": 7105056.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1308 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03246930422919509, "grad_norm": 0.0, "learning_rate": 3.4599999999999995e-07, "loss": 0.0, "num_tokens": 7105954.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1309 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.032494108892471785, "grad_norm": 0.0, "learning_rate": 3.4549999999999996e-07, "loss": 0.0, "num_tokens": 7106928.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2682.0, "completions/max_terminated_length": 2682.0, "completions/mean_length": 2478.5, "completions/mean_terminated_length": 2478.5, "completions/min_length": 2275.0, "completions/min_terminated_length": 2275.0, "epoch": 0.03251891355574848, "grad_norm": 0.0, "learning_rate": 3.45e-07, "loss": 0.0, "num_tokens": 7112777.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1311 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03254371821902518, "grad_norm": 0.0, "learning_rate": 3.4449999999999994e-07, "loss": 0.0, "num_tokens": 7113673.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4060.0, "completions/mean_length": 6126.0, "completions/mean_terminated_length": 4060.0, "completions/min_length": 4060.0, "completions/min_terminated_length": 4060.0, "epoch": 0.03256852288230187, "grad_norm": 3.9304325580596924, "learning_rate": 3.4399999999999996e-07, "loss": -0.707, "num_tokens": 7118637.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1313 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4106.0, "completions/max_terminated_length": 4106.0, "completions/mean_length": 4027.5, "completions/mean_terminated_length": 4027.5, "completions/min_length": 3949.0, "completions/min_terminated_length": 3949.0, "epoch": 0.03259332754557857, "grad_norm": 2.9374167919158936, "learning_rate": 3.435e-07, "loss": -0.0138, "num_tokens": 7127570.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2039.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 2023.5, "completions/mean_terminated_length": 2023.5, "completions/min_length": 2008.0, "completions/min_terminated_length": 2008.0, "epoch": 0.03261813220885527, "grad_norm": 0.0, "learning_rate": 3.43e-07, "loss": 0.0, "num_tokens": 7132529.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5526.0, "completions/max_terminated_length": 5526.0, "completions/mean_length": 4767.0, "completions/mean_terminated_length": 4767.0, "completions/min_length": 4008.0, "completions/min_terminated_length": 4008.0, "epoch": 0.03264293687213196, "grad_norm": 0.0, "learning_rate": 3.425e-07, "loss": 0.0, "num_tokens": 7142993.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 956.0, "completions/max_terminated_length": 956.0, "completions/mean_length": 850.0, "completions/mean_terminated_length": 850.0, "completions/min_length": 744.0, "completions/min_terminated_length": 744.0, "epoch": 0.032667741535408655, "grad_norm": 0.0, "learning_rate": 3.42e-07, "loss": 0.0, "num_tokens": 7145617.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1317 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1714.0, "completions/max_terminated_length": 1714.0, "completions/mean_length": 1481.0, "completions/mean_terminated_length": 1481.0, "completions/min_length": 1248.0, "completions/min_terminated_length": 1248.0, "epoch": 0.032692546198685356, "grad_norm": 0.0, "learning_rate": 3.4150000000000003e-07, "loss": 0.0, "num_tokens": 7149431.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1318 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03271735086196205, "grad_norm": 0.0, "learning_rate": 3.41e-07, "loss": 0.0, "num_tokens": 7150303.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1319 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7763.0, "completions/mean_length": 7977.5, "completions/mean_terminated_length": 7763.0, "completions/min_length": 7763.0, "completions/min_terminated_length": 7763.0, "epoch": 0.03274215552523874, "grad_norm": 0.0, "learning_rate": 3.405e-07, "loss": 0.0, "num_tokens": 7159086.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2421.0, "completions/max_terminated_length": 2421.0, "completions/mean_length": 1682.0, "completions/mean_terminated_length": 1682.0, "completions/min_length": 943.0, "completions/min_terminated_length": 943.0, "epoch": 0.032766960188515444, "grad_norm": 0.0, "learning_rate": 3.4000000000000003e-07, "loss": 0.0, "num_tokens": 7163274.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1321 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5274.0, "completions/max_terminated_length": 5274.0, "completions/mean_length": 4484.0, "completions/mean_terminated_length": 4484.0, "completions/min_length": 3694.0, "completions/min_terminated_length": 3694.0, "epoch": 0.03279176485179214, "grad_norm": 0.0, "learning_rate": 3.395e-07, "loss": 0.0, "num_tokens": 7173164.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1322 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6715.0, "completions/max_terminated_length": 6715.0, "completions/mean_length": 6294.5, "completions/mean_terminated_length": 6294.5, "completions/min_length": 5874.0, "completions/min_terminated_length": 5874.0, "epoch": 0.03281656951506883, "grad_norm": 0.0, "learning_rate": 3.39e-07, "loss": 0.0, "num_tokens": 7186623.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1323 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2960.0, "completions/max_terminated_length": 2960.0, "completions/mean_length": 2157.5, "completions/mean_terminated_length": 2157.5, "completions/min_length": 1355.0, "completions/min_terminated_length": 1355.0, "epoch": 0.03284137417834553, "grad_norm": 0.0, "learning_rate": 3.385e-07, "loss": 0.0, "num_tokens": 7191748.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2527.0, "completions/max_terminated_length": 2527.0, "completions/mean_length": 2097.5, "completions/mean_terminated_length": 2097.5, "completions/min_length": 1668.0, "completions/min_terminated_length": 1668.0, "epoch": 0.032866178841622226, "grad_norm": 3.839759349822998, "learning_rate": 3.38e-07, "loss": -0.1448, "num_tokens": 7196841.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2579.0, "completions/max_terminated_length": 2579.0, "completions/mean_length": 2189.0, "completions/mean_terminated_length": 2189.0, "completions/min_length": 1799.0, "completions/min_terminated_length": 1799.0, "epoch": 0.03289098350489892, "grad_norm": 0.0, "learning_rate": 3.375e-07, "loss": 0.0, "num_tokens": 7202099.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7046.0, "completions/max_terminated_length": 7046.0, "completions/mean_length": 6847.5, "completions/mean_terminated_length": 6847.5, "completions/min_length": 6649.0, "completions/min_terminated_length": 6649.0, "epoch": 0.032915788168175614, "grad_norm": 0.0, "learning_rate": 3.37e-07, "loss": 0.0, "num_tokens": 7216628.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6743.0, "completions/max_terminated_length": 6743.0, "completions/mean_length": 5853.5, "completions/mean_terminated_length": 5853.5, "completions/min_length": 4964.0, "completions/min_terminated_length": 4964.0, "epoch": 0.032940592831452314, "grad_norm": 0.0, "learning_rate": 3.3650000000000003e-07, "loss": 0.0, "num_tokens": 7229259.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1328 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03296539749472901, "grad_norm": 0.0, "learning_rate": 3.36e-07, "loss": 0.0, "num_tokens": 7230111.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1329 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7576.0, "completions/mean_length": 7884.0, "completions/mean_terminated_length": 7576.0, "completions/min_length": 7576.0, "completions/min_terminated_length": 7576.0, "epoch": 0.0329902021580057, "grad_norm": 3.272064685821533, "learning_rate": 3.355e-07, "loss": -0.707, "num_tokens": 7238589.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1330 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3511.0, "completions/max_terminated_length": 3511.0, "completions/mean_length": 2566.0, "completions/mean_terminated_length": 2566.0, "completions/min_length": 1621.0, "completions/min_terminated_length": 1621.0, "epoch": 0.0330150068212824, "grad_norm": 0.0, "learning_rate": 3.35e-07, "loss": 0.0, "num_tokens": 7244691.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1331 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.033039811484559096, "grad_norm": 0.0, "learning_rate": 3.345e-07, "loss": 0.0, "num_tokens": 7245669.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3201.0, "completions/max_terminated_length": 3201.0, "completions/mean_length": 2725.0, "completions/mean_terminated_length": 2725.0, "completions/min_length": 2249.0, "completions/min_terminated_length": 2249.0, "epoch": 0.03306461614783579, "grad_norm": 0.0, "learning_rate": 3.34e-07, "loss": 0.0, "num_tokens": 7252145.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1333 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6915.0, "completions/mean_length": 7553.5, "completions/mean_terminated_length": 6915.0, "completions/min_length": 6915.0, "completions/min_terminated_length": 6915.0, "epoch": 0.03308942081111249, "grad_norm": 0.0, "learning_rate": 3.335e-07, "loss": 0.0, "num_tokens": 7260068.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1334 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5797.0, "completions/mean_length": 6994.5, "completions/mean_terminated_length": 5797.0, "completions/min_length": 5797.0, "completions/min_terminated_length": 5797.0, "epoch": 0.033114225474389185, "grad_norm": 0.0, "learning_rate": 3.33e-07, "loss": 0.0, "num_tokens": 7266877.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1335 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03313903013766588, "grad_norm": 0.0, "learning_rate": 3.325e-07, "loss": 0.0, "num_tokens": 7267819.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3023.0, "completions/max_terminated_length": 3023.0, "completions/mean_length": 2413.0, "completions/mean_terminated_length": 2413.0, "completions/min_length": 1803.0, "completions/min_terminated_length": 1803.0, "epoch": 0.03316383480094258, "grad_norm": 3.3437488079071045, "learning_rate": 3.32e-07, "loss": -0.1787, "num_tokens": 7273461.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2981.0, "completions/max_terminated_length": 2981.0, "completions/mean_length": 2552.5, "completions/mean_terminated_length": 2552.5, "completions/min_length": 2124.0, "completions/min_terminated_length": 2124.0, "epoch": 0.03318863946421927, "grad_norm": 0.0, "learning_rate": 3.315e-07, "loss": 0.0, "num_tokens": 7279512.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1338 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03321344412749597, "grad_norm": 0.0, "learning_rate": 3.31e-07, "loss": 0.0, "num_tokens": 7280386.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7192.0, "completions/mean_length": 7692.0, "completions/mean_terminated_length": 7192.0, "completions/min_length": 7192.0, "completions/min_terminated_length": 7192.0, "epoch": 0.03323824879077267, "grad_norm": 0.0, "learning_rate": 3.305e-07, "loss": 0.0, "num_tokens": 7288556.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7673.0, "completions/max_terminated_length": 7673.0, "completions/mean_length": 6221.0, "completions/mean_terminated_length": 6221.0, "completions/min_length": 4769.0, "completions/min_terminated_length": 4769.0, "epoch": 0.03326305345404936, "grad_norm": 0.0, "learning_rate": 3.3e-07, "loss": 0.0, "num_tokens": 7302150.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7996.0, "completions/max_terminated_length": 7996.0, "completions/mean_length": 6561.5, "completions/mean_terminated_length": 6561.5, "completions/min_length": 5127.0, "completions/min_terminated_length": 5127.0, "epoch": 0.033287858117326055, "grad_norm": 2.1294891834259033, "learning_rate": 3.295e-07, "loss": 0.1546, "num_tokens": 7316207.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3681.0, "completions/max_terminated_length": 3681.0, "completions/mean_length": 3543.5, "completions/mean_terminated_length": 3543.5, "completions/min_length": 3406.0, "completions/min_terminated_length": 3406.0, "epoch": 0.033312662780602756, "grad_norm": 2.9734532833099365, "learning_rate": 3.29e-07, "loss": 0.0274, "num_tokens": 7324212.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5718.0, "completions/mean_length": 6955.0, "completions/mean_terminated_length": 5718.0, "completions/min_length": 5718.0, "completions/min_terminated_length": 5718.0, "epoch": 0.03333746744387945, "grad_norm": 0.0, "learning_rate": 3.285e-07, "loss": 0.0, "num_tokens": 7330766.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5927.0, "completions/max_terminated_length": 5927.0, "completions/mean_length": 4592.5, "completions/mean_terminated_length": 4592.5, "completions/min_length": 3258.0, "completions/min_terminated_length": 3258.0, "epoch": 0.03336227210715614, "grad_norm": 0.0, "learning_rate": 3.28e-07, "loss": 0.0, "num_tokens": 7340949.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2772.0, "completions/max_terminated_length": 2772.0, "completions/mean_length": 2471.0, "completions/mean_terminated_length": 2471.0, "completions/min_length": 2170.0, "completions/min_terminated_length": 2170.0, "epoch": 0.033387076770432844, "grad_norm": 0.0, "learning_rate": 3.275e-07, "loss": 0.0, "num_tokens": 7346795.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1113.0, "completions/max_terminated_length": 1113.0, "completions/mean_length": 989.5, "completions/mean_terminated_length": 989.5, "completions/min_length": 866.0, "completions/min_terminated_length": 866.0, "epoch": 0.03341188143370954, "grad_norm": 0.0, "learning_rate": 3.27e-07, "loss": 0.0, "num_tokens": 7349674.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1347 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03343668609698623, "grad_norm": 0.0, "learning_rate": 3.265e-07, "loss": 0.0, "num_tokens": 7350586.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4219.0, "completions/mean_length": 6205.5, "completions/mean_terminated_length": 4219.0, "completions/min_length": 4219.0, "completions/min_terminated_length": 4219.0, "epoch": 0.03346149076026293, "grad_norm": 0.0, "learning_rate": 3.26e-07, "loss": 0.0, "num_tokens": 7355787.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7140.0, "completions/mean_length": 7666.0, "completions/mean_terminated_length": 7140.0, "completions/min_length": 7140.0, "completions/min_terminated_length": 7140.0, "epoch": 0.033486295423539626, "grad_norm": 3.5712192058563232, "learning_rate": 3.255e-07, "loss": -0.707, "num_tokens": 7363759.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8115.0, "completions/mean_length": 8153.5, "completions/mean_terminated_length": 8115.0, "completions/min_length": 8115.0, "completions/min_terminated_length": 8115.0, "epoch": 0.03351110008681632, "grad_norm": 3.0446126461029053, "learning_rate": 3.25e-07, "loss": -0.707, "num_tokens": 7372772.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1351 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4393.0, "completions/max_terminated_length": 4393.0, "completions/mean_length": 4337.5, "completions/mean_terminated_length": 4337.5, "completions/min_length": 4282.0, "completions/min_terminated_length": 4282.0, "epoch": 0.03353590475009302, "grad_norm": 0.0, "learning_rate": 3.245e-07, "loss": 0.0, "num_tokens": 7382315.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1352 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3904.0, "completions/max_terminated_length": 3904.0, "completions/mean_length": 3022.0, "completions/mean_terminated_length": 3022.0, "completions/min_length": 2140.0, "completions/min_terminated_length": 2140.0, "epoch": 0.033560709413369715, "grad_norm": 0.0, "learning_rate": 3.24e-07, "loss": 0.0, "num_tokens": 7389211.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5575.0, "completions/mean_length": 6883.5, "completions/mean_terminated_length": 5575.0, "completions/min_length": 5575.0, "completions/min_terminated_length": 5575.0, "epoch": 0.03358551407664641, "grad_norm": 0.0, "learning_rate": 3.235e-07, "loss": 0.0, "num_tokens": 7395720.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3425.0, "completions/max_terminated_length": 3425.0, "completions/mean_length": 3386.0, "completions/mean_terminated_length": 3386.0, "completions/min_length": 3347.0, "completions/min_terminated_length": 3347.0, "epoch": 0.0336103187399231, "grad_norm": 0.0, "learning_rate": 3.23e-07, "loss": 0.0, "num_tokens": 7403418.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1355 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0336351234031998, "grad_norm": 0.0, "learning_rate": 3.225e-07, "loss": 0.0, "num_tokens": 7404286.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1356 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4161.0, "completions/max_terminated_length": 4161.0, "completions/mean_length": 3553.5, "completions/mean_terminated_length": 3553.5, "completions/min_length": 2946.0, "completions/min_terminated_length": 2946.0, "epoch": 0.0336599280664765, "grad_norm": 0.0, "learning_rate": 3.22e-07, "loss": 0.0, "num_tokens": 7412259.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1549.0, "completions/mean_length": 4870.5, "completions/mean_terminated_length": 1549.0, "completions/min_length": 1549.0, "completions/min_terminated_length": 1549.0, "epoch": 0.03368473272975319, "grad_norm": 6.59169864654541, "learning_rate": 3.215e-07, "loss": -0.707, "num_tokens": 7414842.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1358 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3289.0, "completions/max_terminated_length": 3289.0, "completions/mean_length": 3269.0, "completions/mean_terminated_length": 3269.0, "completions/min_length": 3249.0, "completions/min_terminated_length": 3249.0, "epoch": 0.03370953739302989, "grad_norm": 0.0, "learning_rate": 3.21e-07, "loss": 0.0, "num_tokens": 7422266.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1359 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.033734342056306585, "grad_norm": 0.0, "learning_rate": 3.205e-07, "loss": 0.0, "num_tokens": 7423148.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6869.0, "completions/max_terminated_length": 6869.0, "completions/mean_length": 4686.5, "completions/mean_terminated_length": 4686.5, "completions/min_length": 2504.0, "completions/min_terminated_length": 2504.0, "epoch": 0.03375914671958328, "grad_norm": 0.0, "learning_rate": 3.2e-07, "loss": 0.0, "num_tokens": 7433347.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1361 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3179.0, "completions/mean_length": 5685.5, "completions/mean_terminated_length": 3179.0, "completions/min_length": 3179.0, "completions/min_terminated_length": 3179.0, "epoch": 0.03378395138285998, "grad_norm": 4.111738204956055, "learning_rate": 3.1949999999999997e-07, "loss": -0.707, "num_tokens": 7437390.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3252.0, "completions/mean_length": 5722.0, "completions/mean_terminated_length": 3252.0, "completions/min_length": 3252.0, "completions/min_terminated_length": 3252.0, "epoch": 0.03380875604613667, "grad_norm": 4.424533367156982, "learning_rate": 3.19e-07, "loss": -0.707, "num_tokens": 7441688.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3622.0, "completions/mean_length": 5907.0, "completions/mean_terminated_length": 3622.0, "completions/min_length": 3622.0, "completions/min_terminated_length": 3622.0, "epoch": 0.03383356070941337, "grad_norm": 5.137767314910889, "learning_rate": 3.185e-07, "loss": -0.707, "num_tokens": 7446184.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7205.0, "completions/mean_length": 7698.5, "completions/mean_terminated_length": 7205.0, "completions/min_length": 7205.0, "completions/min_terminated_length": 7205.0, "epoch": 0.03385836537269007, "grad_norm": 0.0, "learning_rate": 3.18e-07, "loss": 0.0, "num_tokens": 7454259.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1066.0, "completions/max_terminated_length": 1066.0, "completions/mean_length": 883.5, "completions/mean_terminated_length": 883.5, "completions/min_length": 701.0, "completions/min_terminated_length": 701.0, "epoch": 0.03388317003596676, "grad_norm": 0.0, "learning_rate": 3.175e-07, "loss": 0.0, "num_tokens": 7456844.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1366 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.033907974699243455, "grad_norm": 0.0, "learning_rate": 3.17e-07, "loss": 0.0, "num_tokens": 7457740.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1367 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2678.0, "completions/mean_length": 5435.0, "completions/mean_terminated_length": 2678.0, "completions/min_length": 2678.0, "completions/min_terminated_length": 2678.0, "epoch": 0.033932779362520156, "grad_norm": 5.039706707000732, "learning_rate": 3.165e-07, "loss": -0.707, "num_tokens": 7461384.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1368 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3241.0, "completions/mean_length": 5716.5, "completions/mean_terminated_length": 3241.0, "completions/min_length": 3241.0, "completions/min_terminated_length": 3241.0, "epoch": 0.03395758402579685, "grad_norm": 5.119878768920898, "learning_rate": 3.1599999999999997e-07, "loss": -0.707, "num_tokens": 7465465.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2670.0, "completions/max_terminated_length": 2670.0, "completions/mean_length": 2379.5, "completions/mean_terminated_length": 2379.5, "completions/min_length": 2089.0, "completions/min_terminated_length": 2089.0, "epoch": 0.033982388689073544, "grad_norm": 0.0, "learning_rate": 3.155e-07, "loss": 0.0, "num_tokens": 7471072.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1370 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.034007193352350244, "grad_norm": 0.0, "learning_rate": 3.15e-07, "loss": 0.0, "num_tokens": 7472080.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5149.0, "completions/max_terminated_length": 5149.0, "completions/mean_length": 4228.5, "completions/mean_terminated_length": 4228.5, "completions/min_length": 3308.0, "completions/min_terminated_length": 3308.0, "epoch": 0.03403199801562694, "grad_norm": 2.6037657260894775, "learning_rate": 3.1449999999999996e-07, "loss": -0.1539, "num_tokens": 7481423.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1372 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03405680267890363, "grad_norm": 0.0, "learning_rate": 3.14e-07, "loss": 0.0, "num_tokens": 7482281.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1373 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03408160734218033, "grad_norm": 0.0, "learning_rate": 3.135e-07, "loss": 0.0, "num_tokens": 7483165.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1374 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.034106412005457026, "grad_norm": 0.0, "learning_rate": 3.13e-07, "loss": 0.0, "num_tokens": 7484041.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1375 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03413121666873372, "grad_norm": 0.0, "learning_rate": 3.1249999999999997e-07, "loss": 0.0, "num_tokens": 7484867.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6034.0, "completions/mean_length": 7113.0, "completions/mean_terminated_length": 6034.0, "completions/min_length": 6034.0, "completions/min_terminated_length": 6034.0, "epoch": 0.03415602133201042, "grad_norm": 0.0, "learning_rate": 3.12e-07, "loss": 0.0, "num_tokens": 7491843.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1377 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.034180825995287115, "grad_norm": 0.0, "learning_rate": 3.115e-07, "loss": 0.0, "num_tokens": 7492761.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1378 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2070.0, "completions/max_terminated_length": 2070.0, "completions/mean_length": 1757.5, "completions/mean_terminated_length": 1757.5, "completions/min_length": 1445.0, "completions/min_terminated_length": 1445.0, "epoch": 0.03420563065856381, "grad_norm": 0.0, "learning_rate": 3.1099999999999997e-07, "loss": 0.0, "num_tokens": 7497190.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 874.0, "completions/max_terminated_length": 874.0, "completions/mean_length": 832.5, "completions/mean_terminated_length": 832.5, "completions/min_length": 791.0, "completions/min_terminated_length": 791.0, "epoch": 0.03423043532184051, "grad_norm": 0.0, "learning_rate": 3.105e-07, "loss": 0.0, "num_tokens": 7499645.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1454.0, "completions/max_terminated_length": 1454.0, "completions/mean_length": 1327.5, "completions/mean_terminated_length": 1327.5, "completions/min_length": 1201.0, "completions/min_terminated_length": 1201.0, "epoch": 0.0342552399851172, "grad_norm": 0.0, "learning_rate": 3.1e-07, "loss": 0.0, "num_tokens": 7503324.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3123.0, "completions/max_terminated_length": 3123.0, "completions/mean_length": 2984.5, "completions/mean_terminated_length": 2984.5, "completions/min_length": 2846.0, "completions/min_terminated_length": 2846.0, "epoch": 0.0342800446483939, "grad_norm": 0.0, "learning_rate": 3.0949999999999996e-07, "loss": 0.0, "num_tokens": 7510131.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1831.0, "completions/max_terminated_length": 1831.0, "completions/mean_length": 1553.0, "completions/mean_terminated_length": 1553.0, "completions/min_length": 1275.0, "completions/min_terminated_length": 1275.0, "epoch": 0.0343048493116706, "grad_norm": 0.0, "learning_rate": 3.09e-07, "loss": 0.0, "num_tokens": 7514055.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2358.0, "completions/max_terminated_length": 2358.0, "completions/mean_length": 2090.5, "completions/mean_terminated_length": 2090.5, "completions/min_length": 1823.0, "completions/min_terminated_length": 1823.0, "epoch": 0.03432965397494729, "grad_norm": 0.0, "learning_rate": 3.085e-07, "loss": 0.0, "num_tokens": 7519070.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1384 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1034.0, "completions/max_terminated_length": 1034.0, "completions/mean_length": 735.0, "completions/mean_terminated_length": 735.0, "completions/min_length": 436.0, "completions/min_terminated_length": 436.0, "epoch": 0.034354458638223985, "grad_norm": 0.0, "learning_rate": 3.08e-07, "loss": 0.0, "num_tokens": 7521578.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2010.0, "completions/max_terminated_length": 2010.0, "completions/mean_length": 2005.5, "completions/mean_terminated_length": 2005.5, "completions/min_length": 2001.0, "completions/min_terminated_length": 2001.0, "epoch": 0.03437926330150068, "grad_norm": 0.0, "learning_rate": 3.0749999999999997e-07, "loss": 0.0, "num_tokens": 7526393.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5713.0, "completions/max_terminated_length": 5713.0, "completions/mean_length": 4827.0, "completions/mean_terminated_length": 4827.0, "completions/min_length": 3941.0, "completions/min_terminated_length": 3941.0, "epoch": 0.03440406796477738, "grad_norm": 0.0, "learning_rate": 3.07e-07, "loss": 0.0, "num_tokens": 7536841.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1387 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5197.0, "completions/mean_length": 6694.5, "completions/mean_terminated_length": 5197.0, "completions/min_length": 5197.0, "completions/min_terminated_length": 5197.0, "epoch": 0.03442887262805407, "grad_norm": 3.4382307529449463, "learning_rate": 3.065e-07, "loss": -0.707, "num_tokens": 7542968.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1388 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7886.0, "completions/max_terminated_length": 7886.0, "completions/mean_length": 7197.5, "completions/mean_terminated_length": 7197.5, "completions/min_length": 6509.0, "completions/min_terminated_length": 6509.0, "epoch": 0.03445367729133077, "grad_norm": 0.0, "learning_rate": 3.0599999999999996e-07, "loss": 0.0, "num_tokens": 7558217.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2865.0, "completions/max_terminated_length": 2865.0, "completions/mean_length": 2512.5, "completions/mean_terminated_length": 2512.5, "completions/min_length": 2160.0, "completions/min_terminated_length": 2160.0, "epoch": 0.03447848195460747, "grad_norm": 0.0, "learning_rate": 3.055e-07, "loss": 0.0, "num_tokens": 7564134.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1390 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03450328661788416, "grad_norm": 0.0, "learning_rate": 3.05e-07, "loss": 0.0, "num_tokens": 7564986.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1391 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4836.0, "completions/max_terminated_length": 4836.0, "completions/mean_length": 4328.5, "completions/mean_terminated_length": 4328.5, "completions/min_length": 3821.0, "completions/min_terminated_length": 3821.0, "epoch": 0.034528091281160855, "grad_norm": 0.0, "learning_rate": 3.0449999999999995e-07, "loss": 0.0, "num_tokens": 7574529.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3277.0, "completions/max_terminated_length": 3277.0, "completions/mean_length": 3022.5, "completions/mean_terminated_length": 3022.5, "completions/min_length": 2768.0, "completions/min_terminated_length": 2768.0, "epoch": 0.034552895944437556, "grad_norm": 0.0, "learning_rate": 3.0399999999999997e-07, "loss": 0.0, "num_tokens": 7581536.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1393 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03457770060771425, "grad_norm": 0.0, "learning_rate": 3.035e-07, "loss": 0.0, "num_tokens": 7582416.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1394 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.034602505270990944, "grad_norm": 0.0, "learning_rate": 3.03e-07, "loss": 0.0, "num_tokens": 7583448.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5765.0, "completions/max_terminated_length": 5765.0, "completions/mean_length": 4287.0, "completions/mean_terminated_length": 4287.0, "completions/min_length": 2809.0, "completions/min_terminated_length": 2809.0, "epoch": 0.034627309934267644, "grad_norm": 0.0, "learning_rate": 3.0249999999999996e-07, "loss": 0.0, "num_tokens": 7592916.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1474.0, "completions/max_terminated_length": 1474.0, "completions/mean_length": 1112.0, "completions/mean_terminated_length": 1112.0, "completions/min_length": 750.0, "completions/min_terminated_length": 750.0, "epoch": 0.03465211459754434, "grad_norm": 0.0, "learning_rate": 3.02e-07, "loss": 0.0, "num_tokens": 7595940.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1397 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6645.0, "completions/max_terminated_length": 6645.0, "completions/mean_length": 4210.5, "completions/mean_terminated_length": 4210.5, "completions/min_length": 1776.0, "completions/min_terminated_length": 1776.0, "epoch": 0.03467691926082103, "grad_norm": 0.0, "learning_rate": 3.015e-07, "loss": 0.0, "num_tokens": 7605227.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1398 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1107.0, "completions/max_terminated_length": 1107.0, "completions/mean_length": 856.5, "completions/mean_terminated_length": 856.5, "completions/min_length": 606.0, "completions/min_terminated_length": 606.0, "epoch": 0.03470172392409773, "grad_norm": 0.0, "learning_rate": 3.0099999999999996e-07, "loss": 0.0, "num_tokens": 7607816.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1399 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4209.0, "completions/max_terminated_length": 4209.0, "completions/mean_length": 2760.0, "completions/mean_terminated_length": 2760.0, "completions/min_length": 1311.0, "completions/min_terminated_length": 1311.0, "epoch": 0.034726528587374426, "grad_norm": 3.736445426940918, "learning_rate": 3.0049999999999997e-07, "loss": 0.3712, "num_tokens": 7614152.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2022.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1615.5, "completions/mean_terminated_length": 1615.5, "completions/min_length": 1209.0, "completions/min_terminated_length": 1209.0, "epoch": 0.03475133325065112, "grad_norm": 0.0, "learning_rate": 3e-07, "loss": 0.0, "num_tokens": 7618299.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5347.0, "completions/max_terminated_length": 5347.0, "completions/mean_length": 3883.0, "completions/mean_terminated_length": 3883.0, "completions/min_length": 2419.0, "completions/min_terminated_length": 2419.0, "epoch": 0.03477613791392782, "grad_norm": 0.0, "learning_rate": 2.9949999999999995e-07, "loss": 0.0, "num_tokens": 7626893.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5134.0, "completions/max_terminated_length": 5134.0, "completions/mean_length": 3859.5, "completions/mean_terminated_length": 3859.5, "completions/min_length": 2585.0, "completions/min_terminated_length": 2585.0, "epoch": 0.034800942577204515, "grad_norm": 0.0, "learning_rate": 2.9899999999999996e-07, "loss": 0.0, "num_tokens": 7635444.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7482.0, "completions/max_terminated_length": 7482.0, "completions/mean_length": 5817.0, "completions/mean_terminated_length": 5817.0, "completions/min_length": 4152.0, "completions/min_terminated_length": 4152.0, "epoch": 0.03482574724048121, "grad_norm": 0.0, "learning_rate": 2.985e-07, "loss": 0.0, "num_tokens": 7647928.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3004.0, "completions/mean_length": 5598.0, "completions/mean_terminated_length": 3004.0, "completions/min_length": 3004.0, "completions/min_terminated_length": 3004.0, "epoch": 0.03485055190375791, "grad_norm": 0.0, "learning_rate": 2.98e-07, "loss": 0.0, "num_tokens": 7651762.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1405 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0348753565670346, "grad_norm": 0.0, "learning_rate": 2.9749999999999996e-07, "loss": 0.0, "num_tokens": 7652642.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1406 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0349001612303113, "grad_norm": 0.0, "learning_rate": 2.9699999999999997e-07, "loss": 0.0, "num_tokens": 7653604.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1407 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2079.0, "completions/max_terminated_length": 2079.0, "completions/mean_length": 1607.5, "completions/mean_terminated_length": 1607.5, "completions/min_length": 1136.0, "completions/min_terminated_length": 1136.0, "epoch": 0.034924965893588, "grad_norm": 0.0, "learning_rate": 2.965e-07, "loss": 0.0, "num_tokens": 7657655.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2493.0, "completions/max_terminated_length": 2493.0, "completions/mean_length": 1656.5, "completions/mean_terminated_length": 1656.5, "completions/min_length": 820.0, "completions/min_terminated_length": 820.0, "epoch": 0.03494977055686469, "grad_norm": 0.0, "learning_rate": 2.9599999999999995e-07, "loss": 0.0, "num_tokens": 7661764.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1409 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.034974575220141385, "grad_norm": 0.0, "learning_rate": 2.9549999999999997e-07, "loss": 0.0, "num_tokens": 7662746.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3012.0, "completions/max_terminated_length": 3012.0, "completions/mean_length": 2945.0, "completions/mean_terminated_length": 2945.0, "completions/min_length": 2878.0, "completions/min_terminated_length": 2878.0, "epoch": 0.034999379883418086, "grad_norm": 0.0, "learning_rate": 2.95e-07, "loss": 0.0, "num_tokens": 7669568.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1411 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6581.0, "completions/mean_length": 7386.5, "completions/mean_terminated_length": 6581.0, "completions/min_length": 6581.0, "completions/min_terminated_length": 6581.0, "epoch": 0.03502418454669478, "grad_norm": 3.988384246826172, "learning_rate": 2.945e-07, "loss": -0.707, "num_tokens": 7677069.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7747.0, "completions/mean_length": 7969.5, "completions/mean_terminated_length": 7747.0, "completions/min_length": 7747.0, "completions/min_terminated_length": 7747.0, "epoch": 0.03504898920997147, "grad_norm": 0.0, "learning_rate": 2.9399999999999996e-07, "loss": 0.0, "num_tokens": 7685666.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 902.0, "completions/max_terminated_length": 902.0, "completions/mean_length": 730.0, "completions/mean_terminated_length": 730.0, "completions/min_length": 558.0, "completions/min_terminated_length": 558.0, "epoch": 0.035073793873248174, "grad_norm": 0.0, "learning_rate": 2.935e-07, "loss": 0.0, "num_tokens": 7687922.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1478.0, "completions/max_terminated_length": 1478.0, "completions/mean_length": 922.5, "completions/mean_terminated_length": 922.5, "completions/min_length": 367.0, "completions/min_terminated_length": 367.0, "epoch": 0.03509859853652487, "grad_norm": 6.318200588226318, "learning_rate": 2.93e-07, "loss": -0.4257, "num_tokens": 7690675.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5664.0, "completions/max_terminated_length": 5664.0, "completions/mean_length": 4414.5, "completions/mean_terminated_length": 4414.5, "completions/min_length": 3165.0, "completions/min_terminated_length": 3165.0, "epoch": 0.03512340319980156, "grad_norm": 3.4267578125, "learning_rate": 2.9249999999999995e-07, "loss": 0.2001, "num_tokens": 7700394.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2875.0, "completions/max_terminated_length": 2875.0, "completions/mean_length": 2548.0, "completions/mean_terminated_length": 2548.0, "completions/min_length": 2221.0, "completions/min_terminated_length": 2221.0, "epoch": 0.035148207863078255, "grad_norm": 2.9265475273132324, "learning_rate": 2.9199999999999997e-07, "loss": -0.0907, "num_tokens": 7706292.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1417 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4186.0, "completions/max_terminated_length": 4186.0, "completions/mean_length": 3969.5, "completions/mean_terminated_length": 3969.5, "completions/min_length": 3753.0, "completions/min_terminated_length": 3753.0, "epoch": 0.035173012526354956, "grad_norm": 3.4194653034210205, "learning_rate": 2.915e-07, "loss": -0.0386, "num_tokens": 7715097.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6154.0, "completions/max_terminated_length": 6154.0, "completions/mean_length": 3881.0, "completions/mean_terminated_length": 3881.0, "completions/min_length": 1608.0, "completions/min_terminated_length": 1608.0, "epoch": 0.03519781718963165, "grad_norm": 0.0, "learning_rate": 2.9099999999999995e-07, "loss": 0.0, "num_tokens": 7723707.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5752.0, "completions/mean_length": 6972.0, "completions/mean_terminated_length": 5752.0, "completions/min_length": 5752.0, "completions/min_terminated_length": 5752.0, "epoch": 0.035222621852908344, "grad_norm": 3.451873779296875, "learning_rate": 2.9049999999999996e-07, "loss": -0.707, "num_tokens": 7730493.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2109.0, "completions/max_terminated_length": 2109.0, "completions/mean_length": 1668.0, "completions/mean_terminated_length": 1668.0, "completions/min_length": 1227.0, "completions/min_terminated_length": 1227.0, "epoch": 0.035247426516185044, "grad_norm": 0.0, "learning_rate": 2.9e-07, "loss": 0.0, "num_tokens": 7734681.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1421 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1717.0, "completions/mean_length": 4954.5, "completions/mean_terminated_length": 1717.0, "completions/min_length": 1717.0, "completions/min_terminated_length": 1717.0, "epoch": 0.03527223117946174, "grad_norm": 5.740035533905029, "learning_rate": 2.895e-07, "loss": -0.707, "num_tokens": 7737296.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1287.0, "completions/max_terminated_length": 1287.0, "completions/mean_length": 1235.5, "completions/mean_terminated_length": 1235.5, "completions/min_length": 1184.0, "completions/min_terminated_length": 1184.0, "epoch": 0.03529703584273843, "grad_norm": 0.0, "learning_rate": 2.8899999999999995e-07, "loss": 0.0, "num_tokens": 7740681.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1423 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03532184050601513, "grad_norm": 0.0, "learning_rate": 2.8849999999999997e-07, "loss": 0.0, "num_tokens": 7741775.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1647.0, "completions/max_terminated_length": 1647.0, "completions/mean_length": 1403.5, "completions/mean_terminated_length": 1403.5, "completions/min_length": 1160.0, "completions/min_terminated_length": 1160.0, "epoch": 0.03534664516929183, "grad_norm": 0.0, "learning_rate": 2.88e-07, "loss": 0.0, "num_tokens": 7745448.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1425 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03537144983256852, "grad_norm": 0.0, "learning_rate": 2.8749999999999995e-07, "loss": 0.0, "num_tokens": 7746410.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3083.0, "completions/max_terminated_length": 3083.0, "completions/mean_length": 2570.0, "completions/mean_terminated_length": 2570.0, "completions/min_length": 2057.0, "completions/min_terminated_length": 2057.0, "epoch": 0.03539625449584522, "grad_norm": 0.0, "learning_rate": 2.8699999999999996e-07, "loss": 0.0, "num_tokens": 7752398.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5639.0, "completions/max_terminated_length": 5639.0, "completions/mean_length": 3990.0, "completions/mean_terminated_length": 3990.0, "completions/min_length": 2341.0, "completions/min_terminated_length": 2341.0, "epoch": 0.035421059159121915, "grad_norm": 0.0, "learning_rate": 2.865e-07, "loss": 0.0, "num_tokens": 7761264.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2281.0, "completions/max_terminated_length": 2281.0, "completions/mean_length": 1877.0, "completions/mean_terminated_length": 1877.0, "completions/min_length": 1473.0, "completions/min_terminated_length": 1473.0, "epoch": 0.03544586382239861, "grad_norm": 0.0, "learning_rate": 2.8599999999999994e-07, "loss": 0.0, "num_tokens": 7765886.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 4348.5, "completions/mean_terminated_length": 505.0, "completions/min_length": 505.0, "completions/min_terminated_length": 505.0, "epoch": 0.03547066848567531, "grad_norm": 11.742205619812012, "learning_rate": 2.8549999999999996e-07, "loss": -0.707, "num_tokens": 7767179.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7009.0, "completions/max_terminated_length": 7009.0, "completions/mean_length": 6789.5, "completions/mean_terminated_length": 6789.5, "completions/min_length": 6570.0, "completions/min_terminated_length": 6570.0, "epoch": 0.035495473148952, "grad_norm": 0.0, "learning_rate": 2.8499999999999997e-07, "loss": 0.0, "num_tokens": 7781602.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1431 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1536.0, "completions/max_terminated_length": 1536.0, "completions/mean_length": 1491.5, "completions/mean_terminated_length": 1491.5, "completions/min_length": 1447.0, "completions/min_terminated_length": 1447.0, "epoch": 0.0355202778122287, "grad_norm": 4.809178829193115, "learning_rate": 2.845e-07, "loss": -0.0211, "num_tokens": 7785473.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3899.0, "completions/max_terminated_length": 3899.0, "completions/mean_length": 3350.0, "completions/mean_terminated_length": 3350.0, "completions/min_length": 2801.0, "completions/min_terminated_length": 2801.0, "epoch": 0.0355450824755054, "grad_norm": 0.0, "learning_rate": 2.8399999999999995e-07, "loss": 0.0, "num_tokens": 7793023.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2622.0, "completions/max_terminated_length": 2622.0, "completions/mean_length": 2102.0, "completions/mean_terminated_length": 2102.0, "completions/min_length": 1582.0, "completions/min_terminated_length": 1582.0, "epoch": 0.03556988713878209, "grad_norm": 3.288337230682373, "learning_rate": 2.8349999999999996e-07, "loss": -0.1749, "num_tokens": 7798099.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5369.0, "completions/max_terminated_length": 5369.0, "completions/mean_length": 4064.5, "completions/mean_terminated_length": 4064.5, "completions/min_length": 2760.0, "completions/min_terminated_length": 2760.0, "epoch": 0.035594691802058785, "grad_norm": 0.0, "learning_rate": 2.83e-07, "loss": 0.0, "num_tokens": 7807102.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 941.0, "completions/max_terminated_length": 941.0, "completions/mean_length": 849.5, "completions/mean_terminated_length": 849.5, "completions/min_length": 758.0, "completions/min_terminated_length": 758.0, "epoch": 0.035619496465335486, "grad_norm": 0.0, "learning_rate": 2.8249999999999994e-07, "loss": 0.0, "num_tokens": 7809701.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6293.0, "completions/mean_length": 7242.5, "completions/mean_terminated_length": 6293.0, "completions/min_length": 6293.0, "completions/min_terminated_length": 6293.0, "epoch": 0.03564430112861218, "grad_norm": 0.0, "learning_rate": 2.8199999999999996e-07, "loss": 0.0, "num_tokens": 7816830.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1583.0, "completions/max_terminated_length": 1583.0, "completions/mean_length": 1401.5, "completions/mean_terminated_length": 1401.5, "completions/min_length": 1220.0, "completions/min_terminated_length": 1220.0, "epoch": 0.035669105791888873, "grad_norm": 0.0, "learning_rate": 2.8149999999999997e-07, "loss": 0.0, "num_tokens": 7820433.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5822.0, "completions/max_terminated_length": 5822.0, "completions/mean_length": 4933.0, "completions/mean_terminated_length": 4933.0, "completions/min_length": 4044.0, "completions/min_terminated_length": 4044.0, "epoch": 0.035693910455165574, "grad_norm": 0.0, "learning_rate": 2.8100000000000004e-07, "loss": 0.0, "num_tokens": 7831501.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1916.0, "completions/max_terminated_length": 1916.0, "completions/mean_length": 1763.5, "completions/mean_terminated_length": 1763.5, "completions/min_length": 1611.0, "completions/min_terminated_length": 1611.0, "epoch": 0.03571871511844227, "grad_norm": 0.0, "learning_rate": 2.805e-07, "loss": 0.0, "num_tokens": 7835846.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2358.0, "completions/max_terminated_length": 2358.0, "completions/mean_length": 2104.0, "completions/mean_terminated_length": 2104.0, "completions/min_length": 1850.0, "completions/min_terminated_length": 1850.0, "epoch": 0.03574351978171896, "grad_norm": 0.0, "learning_rate": 2.8e-07, "loss": 0.0, "num_tokens": 7840886.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1441 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5017.0, "completions/max_terminated_length": 5017.0, "completions/mean_length": 4923.0, "completions/mean_terminated_length": 4923.0, "completions/min_length": 4829.0, "completions/min_terminated_length": 4829.0, "epoch": 0.03576832444499566, "grad_norm": 2.4047131538391113, "learning_rate": 2.7950000000000003e-07, "loss": -0.0135, "num_tokens": 7851980.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6948.0, "completions/max_terminated_length": 6948.0, "completions/mean_length": 6929.5, "completions/mean_terminated_length": 6929.5, "completions/min_length": 6911.0, "completions/min_terminated_length": 6911.0, "epoch": 0.035793129108272356, "grad_norm": 0.0, "learning_rate": 2.79e-07, "loss": 0.0, "num_tokens": 7866767.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 661.0, "completions/max_terminated_length": 661.0, "completions/mean_length": 615.0, "completions/mean_terminated_length": 615.0, "completions/min_length": 569.0, "completions/min_terminated_length": 569.0, "epoch": 0.03581793377154905, "grad_norm": 0.0, "learning_rate": 2.785e-07, "loss": 0.0, "num_tokens": 7868837.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4626.0, "completions/max_terminated_length": 4626.0, "completions/mean_length": 4418.0, "completions/mean_terminated_length": 4418.0, "completions/min_length": 4210.0, "completions/min_terminated_length": 4210.0, "epoch": 0.035842738434825744, "grad_norm": 0.0, "learning_rate": 2.7800000000000003e-07, "loss": 0.0, "num_tokens": 7878559.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7426.0, "completions/mean_length": 7809.0, "completions/mean_terminated_length": 7426.0, "completions/min_length": 7426.0, "completions/min_terminated_length": 7426.0, "epoch": 0.035867543098102445, "grad_norm": 0.0, "learning_rate": 2.775e-07, "loss": 0.0, "num_tokens": 7886871.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4330.0, "completions/max_terminated_length": 4330.0, "completions/mean_length": 3429.5, "completions/mean_terminated_length": 3429.5, "completions/min_length": 2529.0, "completions/min_terminated_length": 2529.0, "epoch": 0.03589234776137914, "grad_norm": 3.412167549133301, "learning_rate": 2.77e-07, "loss": -0.1856, "num_tokens": 7894550.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3800.0, "completions/max_terminated_length": 3800.0, "completions/mean_length": 3547.5, "completions/mean_terminated_length": 3547.5, "completions/min_length": 3295.0, "completions/min_terminated_length": 3295.0, "epoch": 0.03591715242465583, "grad_norm": 0.0, "learning_rate": 2.765e-07, "loss": 0.0, "num_tokens": 7902503.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1448 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03594195708793253, "grad_norm": 0.0, "learning_rate": 2.7600000000000004e-07, "loss": 0.0, "num_tokens": 7903343.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7036.0, "completions/max_terminated_length": 7036.0, "completions/mean_length": 6628.0, "completions/mean_terminated_length": 6628.0, "completions/min_length": 6220.0, "completions/min_terminated_length": 6220.0, "epoch": 0.03596676175120923, "grad_norm": 0.0, "learning_rate": 2.755e-07, "loss": 0.0, "num_tokens": 7917415.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5627.0, "completions/mean_length": 6909.5, "completions/mean_terminated_length": 5627.0, "completions/min_length": 5627.0, "completions/min_terminated_length": 5627.0, "epoch": 0.03599156641448592, "grad_norm": 3.5906410217285156, "learning_rate": 2.75e-07, "loss": -0.707, "num_tokens": 7923992.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1451 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03601637107776262, "grad_norm": 0.0, "learning_rate": 2.7450000000000003e-07, "loss": 0.0, "num_tokens": 7924892.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3877.0, "completions/max_terminated_length": 3877.0, "completions/mean_length": 2692.0, "completions/mean_terminated_length": 2692.0, "completions/min_length": 1507.0, "completions/min_terminated_length": 1507.0, "epoch": 0.036041175741039315, "grad_norm": 0.0, "learning_rate": 2.74e-07, "loss": 0.0, "num_tokens": 7931114.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1453 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6843.0, "completions/mean_length": 7517.5, "completions/mean_terminated_length": 6843.0, "completions/min_length": 6843.0, "completions/min_terminated_length": 6843.0, "epoch": 0.03606598040431601, "grad_norm": 0.0, "learning_rate": 2.735e-07, "loss": 0.0, "num_tokens": 7939091.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3588.0, "completions/max_terminated_length": 3588.0, "completions/mean_length": 3334.5, "completions/mean_terminated_length": 3334.5, "completions/min_length": 3081.0, "completions/min_terminated_length": 3081.0, "epoch": 0.03609078506759271, "grad_norm": 0.0, "learning_rate": 2.73e-07, "loss": 0.0, "num_tokens": 7946620.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3468.0, "completions/mean_length": 5830.0, "completions/mean_terminated_length": 3468.0, "completions/min_length": 3468.0, "completions/min_terminated_length": 3468.0, "epoch": 0.0361155897308694, "grad_norm": 0.0, "learning_rate": 2.725e-07, "loss": 0.0, "num_tokens": 7950908.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4216.0, "completions/max_terminated_length": 4216.0, "completions/mean_length": 4090.0, "completions/mean_terminated_length": 4090.0, "completions/min_length": 3964.0, "completions/min_terminated_length": 3964.0, "epoch": 0.0361403943941461, "grad_norm": 0.0, "learning_rate": 2.72e-07, "loss": 0.0, "num_tokens": 7959960.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1204.0, "completions/max_terminated_length": 1204.0, "completions/mean_length": 1007.5, "completions/mean_terminated_length": 1007.5, "completions/min_length": 811.0, "completions/min_terminated_length": 811.0, "epoch": 0.0361651990574228, "grad_norm": 0.0, "learning_rate": 2.715e-07, "loss": 0.0, "num_tokens": 7962783.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1458 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03619000372069949, "grad_norm": 0.0, "learning_rate": 2.7100000000000003e-07, "loss": 0.0, "num_tokens": 7963655.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3008.0, "completions/max_terminated_length": 3008.0, "completions/mean_length": 2664.5, "completions/mean_terminated_length": 2664.5, "completions/min_length": 2321.0, "completions/min_terminated_length": 2321.0, "epoch": 0.036214808383976185, "grad_norm": 0.0, "learning_rate": 2.705e-07, "loss": 0.0, "num_tokens": 7969858.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6676.0, "completions/max_terminated_length": 6676.0, "completions/mean_length": 5725.0, "completions/mean_terminated_length": 5725.0, "completions/min_length": 4774.0, "completions/min_terminated_length": 4774.0, "epoch": 0.036239613047252886, "grad_norm": 0.0, "learning_rate": 2.7e-07, "loss": 0.0, "num_tokens": 7982144.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1461 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2204.0, "completions/max_terminated_length": 2204.0, "completions/mean_length": 1611.5, "completions/mean_terminated_length": 1611.5, "completions/min_length": 1019.0, "completions/min_terminated_length": 1019.0, "epoch": 0.03626441771052958, "grad_norm": 0.0, "learning_rate": 2.695e-07, "loss": 0.0, "num_tokens": 7986227.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1462 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5233.0, "completions/mean_length": 6712.5, "completions/mean_terminated_length": 5233.0, "completions/min_length": 5233.0, "completions/min_terminated_length": 5233.0, "epoch": 0.036289222373806274, "grad_norm": 0.0, "learning_rate": 2.69e-07, "loss": 0.0, "num_tokens": 7992454.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1463 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.036314027037082974, "grad_norm": 0.0, "learning_rate": 2.685e-07, "loss": 0.0, "num_tokens": 7993386.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1464 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03633883170035967, "grad_norm": 0.0, "learning_rate": 2.68e-07, "loss": 0.0, "num_tokens": 7994336.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7913.0, "completions/mean_length": 8052.5, "completions/mean_terminated_length": 7913.0, "completions/min_length": 7913.0, "completions/min_terminated_length": 7913.0, "epoch": 0.03636363636363636, "grad_norm": 0.0, "learning_rate": 2.675e-07, "loss": 0.0, "num_tokens": 8003177.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5680.0, "completions/max_terminated_length": 5680.0, "completions/mean_length": 4356.5, "completions/mean_terminated_length": 4356.5, "completions/min_length": 3033.0, "completions/min_terminated_length": 3033.0, "epoch": 0.03638844102691306, "grad_norm": 3.339400053024292, "learning_rate": 2.67e-07, "loss": -0.2148, "num_tokens": 8012728.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1126.0, "completions/max_terminated_length": 1126.0, "completions/mean_length": 982.0, "completions/mean_terminated_length": 982.0, "completions/min_length": 838.0, "completions/min_terminated_length": 838.0, "epoch": 0.036413245690189756, "grad_norm": 0.0, "learning_rate": 2.665e-07, "loss": 0.0, "num_tokens": 8015552.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5097.0, "completions/max_terminated_length": 5097.0, "completions/mean_length": 5054.5, "completions/mean_terminated_length": 5054.5, "completions/min_length": 5012.0, "completions/min_terminated_length": 5012.0, "epoch": 0.03643805035346645, "grad_norm": 0.0, "learning_rate": 2.66e-07, "loss": 0.0, "num_tokens": 8026471.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5830.0, "completions/max_terminated_length": 5830.0, "completions/mean_length": 4314.0, "completions/mean_terminated_length": 4314.0, "completions/min_length": 2798.0, "completions/min_terminated_length": 2798.0, "epoch": 0.03646285501674315, "grad_norm": 0.0, "learning_rate": 2.655e-07, "loss": 0.0, "num_tokens": 8035941.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1470 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6253.0, "completions/max_terminated_length": 6253.0, "completions/mean_length": 5606.0, "completions/mean_terminated_length": 5606.0, "completions/min_length": 4959.0, "completions/min_terminated_length": 4959.0, "epoch": 0.036487659680019845, "grad_norm": 0.0, "learning_rate": 2.65e-07, "loss": 0.0, "num_tokens": 8048031.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1471 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5312.0, "completions/max_terminated_length": 5312.0, "completions/mean_length": 3945.0, "completions/mean_terminated_length": 3945.0, "completions/min_length": 2578.0, "completions/min_terminated_length": 2578.0, "epoch": 0.03651246434329654, "grad_norm": 0.0, "learning_rate": 2.645e-07, "loss": 0.0, "num_tokens": 8056715.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5453.0, "completions/max_terminated_length": 5453.0, "completions/mean_length": 4201.5, "completions/mean_terminated_length": 4201.5, "completions/min_length": 2950.0, "completions/min_terminated_length": 2950.0, "epoch": 0.03653726900657324, "grad_norm": 2.6444976329803467, "learning_rate": 2.64e-07, "loss": 0.2106, "num_tokens": 8066264.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 766.0, "completions/max_terminated_length": 766.0, "completions/mean_length": 687.0, "completions/mean_terminated_length": 687.0, "completions/min_length": 608.0, "completions/min_terminated_length": 608.0, "epoch": 0.03656207366984993, "grad_norm": 0.0, "learning_rate": 2.635e-07, "loss": 0.0, "num_tokens": 8068432.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7596.0, "completions/max_terminated_length": 7596.0, "completions/mean_length": 6352.5, "completions/mean_terminated_length": 6352.5, "completions/min_length": 5109.0, "completions/min_terminated_length": 5109.0, "epoch": 0.03658687833312663, "grad_norm": 0.0, "learning_rate": 2.63e-07, "loss": 0.0, "num_tokens": 8082095.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2523.0, "completions/max_terminated_length": 2523.0, "completions/mean_length": 1409.5, "completions/mean_terminated_length": 1409.5, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "epoch": 0.03661168299640332, "grad_norm": 5.33512020111084, "learning_rate": 2.625e-07, "loss": -0.5585, "num_tokens": 8085728.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1476 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03663648765968002, "grad_norm": 0.0, "learning_rate": 2.62e-07, "loss": 0.0, "num_tokens": 8086720.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5562.0, "completions/max_terminated_length": 5562.0, "completions/mean_length": 5472.0, "completions/mean_terminated_length": 5472.0, "completions/min_length": 5382.0, "completions/min_terminated_length": 5382.0, "epoch": 0.036661292322956715, "grad_norm": 0.0, "learning_rate": 2.615e-07, "loss": 0.0, "num_tokens": 8098958.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1478 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03668609698623341, "grad_norm": 0.0, "learning_rate": 2.61e-07, "loss": 0.0, "num_tokens": 8100010.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 674.0, "completions/max_terminated_length": 674.0, "completions/mean_length": 567.5, "completions/mean_terminated_length": 567.5, "completions/min_length": 461.0, "completions/min_terminated_length": 461.0, "epoch": 0.03671090164951011, "grad_norm": 0.0, "learning_rate": 2.605e-07, "loss": 0.0, "num_tokens": 8101979.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1480 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6653.0, "completions/max_terminated_length": 6653.0, "completions/mean_length": 5506.5, "completions/mean_terminated_length": 5506.5, "completions/min_length": 4360.0, "completions/min_terminated_length": 4360.0, "epoch": 0.0367357063127868, "grad_norm": 0.0, "learning_rate": 2.6e-07, "loss": 0.0, "num_tokens": 8113846.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1481 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1764.0, "completions/max_terminated_length": 1764.0, "completions/mean_length": 1700.0, "completions/mean_terminated_length": 1700.0, "completions/min_length": 1636.0, "completions/min_terminated_length": 1636.0, "epoch": 0.0367605109760635, "grad_norm": 0.0, "learning_rate": 2.595e-07, "loss": 0.0, "num_tokens": 8118070.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1482 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0367853156393402, "grad_norm": 0.0, "learning_rate": 2.59e-07, "loss": 0.0, "num_tokens": 8118970.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 806.0, "completions/max_terminated_length": 806.0, "completions/mean_length": 766.5, "completions/mean_terminated_length": 766.5, "completions/min_length": 727.0, "completions/min_terminated_length": 727.0, "epoch": 0.03681012030261689, "grad_norm": 0.0, "learning_rate": 2.585e-07, "loss": 0.0, "num_tokens": 8121293.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6763.0, "completions/max_terminated_length": 6763.0, "completions/mean_length": 6496.0, "completions/mean_terminated_length": 6496.0, "completions/min_length": 6229.0, "completions/min_terminated_length": 6229.0, "epoch": 0.036834924965893585, "grad_norm": 0.0, "learning_rate": 2.58e-07, "loss": 0.0, "num_tokens": 8135099.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1485 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.036859729629170286, "grad_norm": 0.0, "learning_rate": 2.5749999999999997e-07, "loss": 0.0, "num_tokens": 8136123.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3586.0, "completions/mean_length": 5889.0, "completions/mean_terminated_length": 3586.0, "completions/min_length": 3586.0, "completions/min_terminated_length": 3586.0, "epoch": 0.03688453429244698, "grad_norm": 3.9844236373901367, "learning_rate": 2.57e-07, "loss": -0.707, "num_tokens": 8140689.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1487 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5252.0, "completions/mean_length": 6722.0, "completions/mean_terminated_length": 5252.0, "completions/min_length": 5252.0, "completions/min_terminated_length": 5252.0, "epoch": 0.036909338955723674, "grad_norm": 3.5087060928344727, "learning_rate": 2.565e-07, "loss": -0.707, "num_tokens": 8146787.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1488 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2255.0, "completions/mean_length": 5223.5, "completions/mean_terminated_length": 2255.0, "completions/min_length": 2255.0, "completions/min_terminated_length": 2255.0, "epoch": 0.036934143619000374, "grad_norm": 4.642655849456787, "learning_rate": 2.56e-07, "loss": -0.707, "num_tokens": 8149862.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1489 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2861.0, "completions/mean_length": 5526.5, "completions/mean_terminated_length": 2861.0, "completions/min_length": 2861.0, "completions/min_terminated_length": 2861.0, "epoch": 0.03695894828227707, "grad_norm": 0.0, "learning_rate": 2.555e-07, "loss": 0.0, "num_tokens": 8153583.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6286.0, "completions/mean_length": 7239.0, "completions/mean_terminated_length": 6286.0, "completions/min_length": 6286.0, "completions/min_terminated_length": 6286.0, "epoch": 0.03698375294555376, "grad_norm": 0.0, "learning_rate": 2.55e-07, "loss": 0.0, "num_tokens": 8160795.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2407.0, "completions/max_terminated_length": 2407.0, "completions/mean_length": 1842.0, "completions/mean_terminated_length": 1842.0, "completions/min_length": 1277.0, "completions/min_terminated_length": 1277.0, "epoch": 0.03700855760883046, "grad_norm": 4.634438514709473, "learning_rate": 2.545e-07, "loss": 0.2169, "num_tokens": 8165371.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6849.0, "completions/mean_length": 7520.5, "completions/mean_terminated_length": 6849.0, "completions/min_length": 6849.0, "completions/min_terminated_length": 6849.0, "epoch": 0.037033362272107156, "grad_norm": 3.528571605682373, "learning_rate": 2.5399999999999997e-07, "loss": -0.707, "num_tokens": 8173236.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1493 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03705816693538385, "grad_norm": 0.0, "learning_rate": 2.535e-07, "loss": 0.0, "num_tokens": 8174150.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1494 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03708297159866055, "grad_norm": 0.0, "learning_rate": 2.53e-07, "loss": 0.0, "num_tokens": 8175044.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6551.0, "completions/max_terminated_length": 6551.0, "completions/mean_length": 4070.0, "completions/mean_terminated_length": 4070.0, "completions/min_length": 1589.0, "completions/min_terminated_length": 1589.0, "epoch": 0.037107776261937245, "grad_norm": 0.0, "learning_rate": 2.5249999999999996e-07, "loss": 0.0, "num_tokens": 8184028.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1565.0, "completions/max_terminated_length": 1565.0, "completions/mean_length": 1363.5, "completions/mean_terminated_length": 1363.5, "completions/min_length": 1162.0, "completions/min_terminated_length": 1162.0, "epoch": 0.03713258092521394, "grad_norm": 0.0, "learning_rate": 2.52e-07, "loss": 0.0, "num_tokens": 8187629.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3476.0, "completions/max_terminated_length": 3476.0, "completions/mean_length": 2546.5, "completions/mean_terminated_length": 2546.5, "completions/min_length": 1617.0, "completions/min_terminated_length": 1617.0, "epoch": 0.03715738558849064, "grad_norm": 0.0, "learning_rate": 2.515e-07, "loss": 0.0, "num_tokens": 8193526.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2001.0, "completions/max_terminated_length": 2001.0, "completions/mean_length": 1637.0, "completions/mean_terminated_length": 1637.0, "completions/min_length": 1273.0, "completions/min_terminated_length": 1273.0, "epoch": 0.03718219025176733, "grad_norm": 0.0, "learning_rate": 2.51e-07, "loss": 0.0, "num_tokens": 8197640.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1499 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2235.0, "completions/max_terminated_length": 2235.0, "completions/mean_length": 1647.5, "completions/mean_terminated_length": 1647.5, "completions/min_length": 1060.0, "completions/min_terminated_length": 1060.0, "epoch": 0.03720699491504403, "grad_norm": 0.0, "learning_rate": 2.5049999999999997e-07, "loss": 0.0, "num_tokens": 8201997.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3368.0, "completions/mean_length": 5780.0, "completions/mean_terminated_length": 3368.0, "completions/min_length": 3368.0, "completions/min_terminated_length": 3368.0, "epoch": 0.03723179957832073, "grad_norm": 0.0, "learning_rate": 2.5e-07, "loss": 0.0, "num_tokens": 8206247.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1501 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6630.0, "completions/max_terminated_length": 6630.0, "completions/mean_length": 5323.5, "completions/mean_terminated_length": 5323.5, "completions/min_length": 4017.0, "completions/min_terminated_length": 4017.0, "epoch": 0.03725660424159742, "grad_norm": 0.0, "learning_rate": 2.495e-07, "loss": 0.0, "num_tokens": 8217826.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1502 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1411.0, "completions/max_terminated_length": 1411.0, "completions/mean_length": 1070.5, "completions/mean_terminated_length": 1070.5, "completions/min_length": 730.0, "completions/min_terminated_length": 730.0, "epoch": 0.037281408904874115, "grad_norm": 0.0, "learning_rate": 2.4899999999999997e-07, "loss": 0.0, "num_tokens": 8220789.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1503 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03730621356815081, "grad_norm": 0.0, "learning_rate": 2.485e-07, "loss": 0.0, "num_tokens": 8221653.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1504 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03733101823142751, "grad_norm": 0.0, "learning_rate": 2.48e-07, "loss": 0.0, "num_tokens": 8222525.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4483.0, "completions/mean_length": 6337.5, "completions/mean_terminated_length": 4483.0, "completions/min_length": 4483.0, "completions/min_terminated_length": 4483.0, "epoch": 0.0373558228947042, "grad_norm": 0.0, "learning_rate": 2.475e-07, "loss": 0.0, "num_tokens": 8227864.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3771.0, "completions/max_terminated_length": 3771.0, "completions/mean_length": 3364.0, "completions/mean_terminated_length": 3364.0, "completions/min_length": 2957.0, "completions/min_terminated_length": 2957.0, "epoch": 0.0373806275579809, "grad_norm": 3.597874641418457, "learning_rate": 2.47e-07, "loss": 0.0855, "num_tokens": 8235422.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1507 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5400.0, "completions/max_terminated_length": 5400.0, "completions/mean_length": 4311.0, "completions/mean_terminated_length": 4311.0, "completions/min_length": 3222.0, "completions/min_terminated_length": 3222.0, "epoch": 0.0374054322212576, "grad_norm": 0.0, "learning_rate": 2.465e-07, "loss": 0.0, "num_tokens": 8244930.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1723.0, "completions/max_terminated_length": 1723.0, "completions/mean_length": 1502.0, "completions/mean_terminated_length": 1502.0, "completions/min_length": 1281.0, "completions/min_terminated_length": 1281.0, "epoch": 0.03743023688453429, "grad_norm": 0.0, "learning_rate": 2.46e-07, "loss": 0.0, "num_tokens": 8248832.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1509 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.037455041547810985, "grad_norm": 0.0, "learning_rate": 2.4549999999999997e-07, "loss": 0.0, "num_tokens": 8249694.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1510 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.037479846211087686, "grad_norm": 0.0, "learning_rate": 2.45e-07, "loss": 0.0, "num_tokens": 8250634.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1511 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3406.0, "completions/max_terminated_length": 3406.0, "completions/mean_length": 2297.0, "completions/mean_terminated_length": 2297.0, "completions/min_length": 1188.0, "completions/min_terminated_length": 1188.0, "epoch": 0.03750465087436438, "grad_norm": 0.0, "learning_rate": 2.445e-07, "loss": 0.0, "num_tokens": 8256072.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1512 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.037529455537641074, "grad_norm": 0.0, "learning_rate": 2.4399999999999996e-07, "loss": 0.0, "num_tokens": 8257292.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1513 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.037554260200917775, "grad_norm": 0.0, "learning_rate": 2.435e-07, "loss": 0.0, "num_tokens": 8258350.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2284.0, "completions/max_terminated_length": 2284.0, "completions/mean_length": 1836.0, "completions/mean_terminated_length": 1836.0, "completions/min_length": 1388.0, "completions/min_terminated_length": 1388.0, "epoch": 0.03757906486419447, "grad_norm": 0.0, "learning_rate": 2.43e-07, "loss": 0.0, "num_tokens": 8262912.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1515 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03760386952747116, "grad_norm": 0.0, "learning_rate": 2.425e-07, "loss": 0.0, "num_tokens": 8263938.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5067.0, "completions/mean_length": 6629.5, "completions/mean_terminated_length": 5067.0, "completions/min_length": 5067.0, "completions/min_terminated_length": 5067.0, "epoch": 0.03762867419074786, "grad_norm": 0.0, "learning_rate": 2.4199999999999997e-07, "loss": 0.0, "num_tokens": 8269823.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1517 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03765347885402456, "grad_norm": 0.0, "learning_rate": 2.415e-07, "loss": 0.0, "num_tokens": 8271017.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1518 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03767828351730125, "grad_norm": 0.0, "learning_rate": 2.41e-07, "loss": 0.0, "num_tokens": 8271901.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1519 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3739.0, "completions/max_terminated_length": 3739.0, "completions/mean_length": 2841.5, "completions/mean_terminated_length": 2841.5, "completions/min_length": 1944.0, "completions/min_terminated_length": 1944.0, "epoch": 0.03770308818057795, "grad_norm": 0.0, "learning_rate": 2.4049999999999996e-07, "loss": 0.0, "num_tokens": 8278498.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1520 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1609.0, "completions/max_terminated_length": 1609.0, "completions/mean_length": 1576.5, "completions/mean_terminated_length": 1576.5, "completions/min_length": 1544.0, "completions/min_terminated_length": 1544.0, "epoch": 0.037727892843854645, "grad_norm": 0.0, "learning_rate": 2.4e-07, "loss": 0.0, "num_tokens": 8282499.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1521 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5409.0, "completions/max_terminated_length": 5409.0, "completions/mean_length": 4033.5, "completions/mean_terminated_length": 4033.5, "completions/min_length": 2658.0, "completions/min_terminated_length": 2658.0, "epoch": 0.03775269750713134, "grad_norm": 0.0, "learning_rate": 2.395e-07, "loss": 0.0, "num_tokens": 8291580.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1522 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03777750217040804, "grad_norm": 0.0, "learning_rate": 2.3899999999999996e-07, "loss": 0.0, "num_tokens": 8292446.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5886.0, "completions/max_terminated_length": 5886.0, "completions/mean_length": 5859.5, "completions/mean_terminated_length": 5859.5, "completions/min_length": 5833.0, "completions/min_terminated_length": 5833.0, "epoch": 0.03780230683368473, "grad_norm": 0.0, "learning_rate": 2.3849999999999997e-07, "loss": 0.0, "num_tokens": 8305119.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1524 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03782711149696143, "grad_norm": 0.0, "learning_rate": 2.38e-07, "loss": 0.0, "num_tokens": 8306401.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4672.0, "completions/max_terminated_length": 4672.0, "completions/mean_length": 4327.5, "completions/mean_terminated_length": 4327.5, "completions/min_length": 3983.0, "completions/min_terminated_length": 3983.0, "epoch": 0.03785191616023813, "grad_norm": 0.0, "learning_rate": 2.3749999999999998e-07, "loss": 0.0, "num_tokens": 8315978.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4633.0, "completions/max_terminated_length": 4633.0, "completions/mean_length": 4185.5, "completions/mean_terminated_length": 4185.5, "completions/min_length": 3738.0, "completions/min_terminated_length": 3738.0, "epoch": 0.03787672082351482, "grad_norm": 3.038404941558838, "learning_rate": 2.3699999999999996e-07, "loss": -0.0756, "num_tokens": 8325207.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1527 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5067.0, "completions/max_terminated_length": 5067.0, "completions/mean_length": 4711.5, "completions/mean_terminated_length": 4711.5, "completions/min_length": 4356.0, "completions/min_terminated_length": 4356.0, "epoch": 0.037901525486791515, "grad_norm": 0.0, "learning_rate": 2.3649999999999998e-07, "loss": 0.0, "num_tokens": 8335538.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1528 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.037926330150068216, "grad_norm": 0.0, "learning_rate": 2.3599999999999997e-07, "loss": 0.0, "num_tokens": 8336454.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6211.0, "completions/mean_length": 7201.5, "completions/mean_terminated_length": 6211.0, "completions/min_length": 6211.0, "completions/min_terminated_length": 6211.0, "epoch": 0.03795113481334491, "grad_norm": 3.2669172286987305, "learning_rate": 2.3549999999999998e-07, "loss": -0.707, "num_tokens": 8343601.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1530 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.037975939476621604, "grad_norm": 0.0, "learning_rate": 2.3499999999999997e-07, "loss": 0.0, "num_tokens": 8344533.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5808.0, "completions/mean_length": 7000.0, "completions/mean_terminated_length": 5808.0, "completions/min_length": 5808.0, "completions/min_terminated_length": 5808.0, "epoch": 0.038000744139898304, "grad_norm": 0.0, "learning_rate": 2.3449999999999996e-07, "loss": 0.0, "num_tokens": 8351481.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1532 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.038025548803175, "grad_norm": 0.0, "learning_rate": 2.34e-07, "loss": 0.0, "num_tokens": 8352391.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1533 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03805035346645169, "grad_norm": 0.0, "learning_rate": 2.335e-07, "loss": 0.0, "num_tokens": 8353339.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6969.0, "completions/max_terminated_length": 6969.0, "completions/mean_length": 6663.0, "completions/mean_terminated_length": 6663.0, "completions/min_length": 6357.0, "completions/min_terminated_length": 6357.0, "epoch": 0.038075158129728386, "grad_norm": 0.0, "learning_rate": 2.33e-07, "loss": 0.0, "num_tokens": 8367569.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4723.0, "completions/max_terminated_length": 4723.0, "completions/mean_length": 4441.0, "completions/mean_terminated_length": 4441.0, "completions/min_length": 4159.0, "completions/min_terminated_length": 4159.0, "epoch": 0.038099962793005086, "grad_norm": 0.0, "learning_rate": 2.325e-07, "loss": 0.0, "num_tokens": 8377533.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3345.0, "completions/max_terminated_length": 3345.0, "completions/mean_length": 2657.0, "completions/mean_terminated_length": 2657.0, "completions/min_length": 1969.0, "completions/min_terminated_length": 1969.0, "epoch": 0.03812476745628178, "grad_norm": 0.0, "learning_rate": 2.32e-07, "loss": 0.0, "num_tokens": 8383705.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1537 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6816.0, "completions/max_terminated_length": 6816.0, "completions/mean_length": 5913.0, "completions/mean_terminated_length": 5913.0, "completions/min_length": 5010.0, "completions/min_terminated_length": 5010.0, "epoch": 0.038149572119558474, "grad_norm": 0.0, "learning_rate": 2.315e-07, "loss": 0.0, "num_tokens": 8396471.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3167.0, "completions/max_terminated_length": 3167.0, "completions/mean_length": 3149.0, "completions/mean_terminated_length": 3149.0, "completions/min_length": 3131.0, "completions/min_terminated_length": 3131.0, "epoch": 0.038174376782835175, "grad_norm": 0.0, "learning_rate": 2.31e-07, "loss": 0.0, "num_tokens": 8403689.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1539 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5230.0, "completions/max_terminated_length": 5230.0, "completions/mean_length": 4210.0, "completions/mean_terminated_length": 4210.0, "completions/min_length": 3190.0, "completions/min_terminated_length": 3190.0, "epoch": 0.03819918144611187, "grad_norm": 0.0, "learning_rate": 2.305e-07, "loss": 0.0, "num_tokens": 8413087.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5371.0, "completions/max_terminated_length": 5371.0, "completions/mean_length": 4397.5, "completions/mean_terminated_length": 4397.5, "completions/min_length": 3424.0, "completions/min_terminated_length": 3424.0, "epoch": 0.03822398610938856, "grad_norm": 0.0, "learning_rate": 2.3e-07, "loss": 0.0, "num_tokens": 8422710.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1541 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7529.0, "completions/max_terminated_length": 7529.0, "completions/mean_length": 5087.5, "completions/mean_terminated_length": 5087.5, "completions/min_length": 2646.0, "completions/min_terminated_length": 2646.0, "epoch": 0.03824879077266526, "grad_norm": 2.7185306549072266, "learning_rate": 2.295e-07, "loss": 0.3393, "num_tokens": 8433875.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1542 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6486.0, "completions/max_terminated_length": 6486.0, "completions/mean_length": 4963.0, "completions/mean_terminated_length": 4963.0, "completions/min_length": 3440.0, "completions/min_terminated_length": 3440.0, "epoch": 0.03827359543594196, "grad_norm": 3.049461603164673, "learning_rate": 2.29e-07, "loss": 0.217, "num_tokens": 8444625.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1543 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4865.0, "completions/max_terminated_length": 4865.0, "completions/mean_length": 3909.5, "completions/mean_terminated_length": 3909.5, "completions/min_length": 2954.0, "completions/min_terminated_length": 2954.0, "epoch": 0.03829840009921865, "grad_norm": 3.008287191390991, "learning_rate": 2.285e-07, "loss": 0.1728, "num_tokens": 8453460.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6667.0, "completions/mean_length": 7429.5, "completions/mean_terminated_length": 6667.0, "completions/min_length": 6667.0, "completions/min_terminated_length": 6667.0, "epoch": 0.03832320476249535, "grad_norm": 3.456324338912964, "learning_rate": 2.28e-07, "loss": -0.707, "num_tokens": 8461165.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7680.0, "completions/max_terminated_length": 7680.0, "completions/mean_length": 6048.0, "completions/mean_terminated_length": 6048.0, "completions/min_length": 4416.0, "completions/min_terminated_length": 4416.0, "epoch": 0.038348009425772045, "grad_norm": 2.758958101272583, "learning_rate": 2.275e-07, "loss": 0.1908, "num_tokens": 8474251.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1546 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2143.0, "completions/max_terminated_length": 2143.0, "completions/mean_length": 1829.0, "completions/mean_terminated_length": 1829.0, "completions/min_length": 1515.0, "completions/min_terminated_length": 1515.0, "epoch": 0.03837281408904874, "grad_norm": 0.0, "learning_rate": 2.27e-07, "loss": 0.0, "num_tokens": 8478799.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7773.0, "completions/max_terminated_length": 7773.0, "completions/mean_length": 6489.0, "completions/mean_terminated_length": 6489.0, "completions/min_length": 5205.0, "completions/min_terminated_length": 5205.0, "epoch": 0.03839761875232544, "grad_norm": 2.743303060531616, "learning_rate": 2.265e-07, "loss": 0.1399, "num_tokens": 8492615.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7840.0, "completions/mean_length": 8016.0, "completions/mean_terminated_length": 7840.0, "completions/min_length": 7840.0, "completions/min_terminated_length": 7840.0, "epoch": 0.03842242341560213, "grad_norm": 3.0860493183135986, "learning_rate": 2.2599999999999999e-07, "loss": -0.707, "num_tokens": 8501387.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2366.0, "completions/mean_length": 5279.0, "completions/mean_terminated_length": 2366.0, "completions/min_length": 2366.0, "completions/min_terminated_length": 2366.0, "epoch": 0.03844722807887883, "grad_norm": 6.328463077545166, "learning_rate": 2.255e-07, "loss": -0.707, "num_tokens": 8504675.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1550 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 691.0, "completions/max_terminated_length": 691.0, "completions/mean_length": 597.0, "completions/mean_terminated_length": 597.0, "completions/min_length": 503.0, "completions/min_terminated_length": 503.0, "epoch": 0.03847203274215553, "grad_norm": 0.0, "learning_rate": 2.25e-07, "loss": 0.0, "num_tokens": 8506705.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5839.0, "completions/max_terminated_length": 5839.0, "completions/mean_length": 5559.0, "completions/mean_terminated_length": 5559.0, "completions/min_length": 5279.0, "completions/min_terminated_length": 5279.0, "epoch": 0.03849683740543222, "grad_norm": 2.4593844413757324, "learning_rate": 2.245e-07, "loss": 0.0356, "num_tokens": 8518753.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1552 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2240.0, "completions/max_terminated_length": 2240.0, "completions/mean_length": 1777.5, "completions/mean_terminated_length": 1777.5, "completions/min_length": 1315.0, "completions/min_terminated_length": 1315.0, "epoch": 0.038521642068708915, "grad_norm": 0.0, "learning_rate": 2.24e-07, "loss": 0.0, "num_tokens": 8523238.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3678.0, "completions/max_terminated_length": 3678.0, "completions/mean_length": 3354.5, "completions/mean_terminated_length": 3354.5, "completions/min_length": 3031.0, "completions/min_terminated_length": 3031.0, "epoch": 0.038546446731985616, "grad_norm": 0.0, "learning_rate": 2.2349999999999998e-07, "loss": 0.0, "num_tokens": 8530759.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1076.0, "completions/max_terminated_length": 1076.0, "completions/mean_length": 1061.5, "completions/mean_terminated_length": 1061.5, "completions/min_length": 1047.0, "completions/min_terminated_length": 1047.0, "epoch": 0.03857125139526231, "grad_norm": 0.0, "learning_rate": 2.23e-07, "loss": 0.0, "num_tokens": 8533690.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1555 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.038596056058539004, "grad_norm": 0.0, "learning_rate": 2.225e-07, "loss": 0.0, "num_tokens": 8534574.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7567.0, "completions/max_terminated_length": 7567.0, "completions/mean_length": 6541.0, "completions/mean_terminated_length": 6541.0, "completions/min_length": 5515.0, "completions/min_terminated_length": 5515.0, "epoch": 0.038620860721815704, "grad_norm": 0.0, "learning_rate": 2.22e-07, "loss": 0.0, "num_tokens": 8548510.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1257.0, "completions/max_terminated_length": 1257.0, "completions/mean_length": 1212.5, "completions/mean_terminated_length": 1212.5, "completions/min_length": 1168.0, "completions/min_terminated_length": 1168.0, "epoch": 0.0386456653850924, "grad_norm": 0.0, "learning_rate": 2.215e-07, "loss": 0.0, "num_tokens": 8551871.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 962.0, "completions/max_terminated_length": 962.0, "completions/mean_length": 780.0, "completions/mean_terminated_length": 780.0, "completions/min_length": 598.0, "completions/min_terminated_length": 598.0, "epoch": 0.03867047004836909, "grad_norm": 0.0, "learning_rate": 2.2099999999999998e-07, "loss": 0.0, "num_tokens": 8554303.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1559 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 557.0, "completions/max_terminated_length": 557.0, "completions/mean_length": 533.0, "completions/mean_terminated_length": 533.0, "completions/min_length": 509.0, "completions/min_terminated_length": 509.0, "epoch": 0.03869527471164579, "grad_norm": 0.0, "learning_rate": 2.205e-07, "loss": 0.0, "num_tokens": 8556197.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 625.0, "completions/max_terminated_length": 625.0, "completions/mean_length": 556.0, "completions/mean_terminated_length": 556.0, "completions/min_length": 487.0, "completions/min_terminated_length": 487.0, "epoch": 0.038720079374922486, "grad_norm": 0.0, "learning_rate": 2.1999999999999998e-07, "loss": 0.0, "num_tokens": 8558117.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1561 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6421.0, "completions/mean_length": 7306.5, "completions/mean_terminated_length": 6421.0, "completions/min_length": 6421.0, "completions/min_terminated_length": 6421.0, "epoch": 0.03874488403819918, "grad_norm": 3.3694405555725098, "learning_rate": 2.195e-07, "loss": -0.707, "num_tokens": 8565460.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1562 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2861.0, "completions/max_terminated_length": 2861.0, "completions/mean_length": 2320.5, "completions/mean_terminated_length": 2320.5, "completions/min_length": 1780.0, "completions/min_terminated_length": 1780.0, "epoch": 0.03876968870147588, "grad_norm": 0.0, "learning_rate": 2.19e-07, "loss": 0.0, "num_tokens": 8570939.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5140.0, "completions/max_terminated_length": 5140.0, "completions/mean_length": 4851.5, "completions/mean_terminated_length": 4851.5, "completions/min_length": 4563.0, "completions/min_terminated_length": 4563.0, "epoch": 0.038794493364752575, "grad_norm": 0.0, "learning_rate": 2.1849999999999998e-07, "loss": 0.0, "num_tokens": 8581438.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1564 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03881929802802927, "grad_norm": 0.0, "learning_rate": 2.18e-07, "loss": 0.0, "num_tokens": 8582416.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1565 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03884410269130596, "grad_norm": 0.0, "learning_rate": 2.1749999999999998e-07, "loss": 0.0, "num_tokens": 8583414.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1566 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7317.0, "completions/mean_length": 7754.5, "completions/mean_terminated_length": 7317.0, "completions/min_length": 7317.0, "completions/min_terminated_length": 7317.0, "epoch": 0.03886890735458266, "grad_norm": 3.284874200820923, "learning_rate": 2.17e-07, "loss": -0.707, "num_tokens": 8591581.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1567 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03889371201785936, "grad_norm": 0.0, "learning_rate": 2.1649999999999999e-07, "loss": 0.0, "num_tokens": 8592459.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1568 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3423.0, "completions/mean_length": 5807.5, "completions/mean_terminated_length": 3423.0, "completions/min_length": 3423.0, "completions/min_terminated_length": 3423.0, "epoch": 0.03891851668113605, "grad_norm": 4.51422119140625, "learning_rate": 2.1599999999999998e-07, "loss": -0.707, "num_tokens": 8596958.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3748.0, "completions/max_terminated_length": 3748.0, "completions/mean_length": 2757.0, "completions/mean_terminated_length": 2757.0, "completions/min_length": 1766.0, "completions/min_terminated_length": 1766.0, "epoch": 0.03894332134441275, "grad_norm": 0.0, "learning_rate": 2.155e-07, "loss": 0.0, "num_tokens": 8603370.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7480.0, "completions/max_terminated_length": 7480.0, "completions/mean_length": 6361.0, "completions/mean_terminated_length": 6361.0, "completions/min_length": 5242.0, "completions/min_terminated_length": 5242.0, "epoch": 0.038968126007689445, "grad_norm": 0.0, "learning_rate": 2.1499999999999998e-07, "loss": 0.0, "num_tokens": 8617136.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5282.0, "completions/max_terminated_length": 5282.0, "completions/mean_length": 4820.5, "completions/mean_terminated_length": 4820.5, "completions/min_length": 4359.0, "completions/min_terminated_length": 4359.0, "epoch": 0.03899293067096614, "grad_norm": 3.22861909866333, "learning_rate": 2.145e-07, "loss": -0.0677, "num_tokens": 8627629.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1572 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3865.0, "completions/max_terminated_length": 3865.0, "completions/mean_length": 3186.5, "completions/mean_terminated_length": 3186.5, "completions/min_length": 2508.0, "completions/min_terminated_length": 2508.0, "epoch": 0.03901773533424284, "grad_norm": 3.003162384033203, "learning_rate": 2.1399999999999998e-07, "loss": -0.1505, "num_tokens": 8635262.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7984.0, "completions/max_terminated_length": 7984.0, "completions/mean_length": 7804.0, "completions/mean_terminated_length": 7804.0, "completions/min_length": 7624.0, "completions/min_terminated_length": 7624.0, "epoch": 0.03904253999751953, "grad_norm": 2.7962746620178223, "learning_rate": 2.1349999999999997e-07, "loss": -0.0163, "num_tokens": 8651836.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2470.0, "completions/max_terminated_length": 2470.0, "completions/mean_length": 2225.0, "completions/mean_terminated_length": 2225.0, "completions/min_length": 1980.0, "completions/min_terminated_length": 1980.0, "epoch": 0.03906734466079623, "grad_norm": 0.0, "learning_rate": 2.13e-07, "loss": 0.0, "num_tokens": 8657510.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6177.0, "completions/max_terminated_length": 6177.0, "completions/mean_length": 6088.0, "completions/mean_terminated_length": 6088.0, "completions/min_length": 5999.0, "completions/min_terminated_length": 5999.0, "epoch": 0.03909214932407293, "grad_norm": 0.0, "learning_rate": 2.1249999999999998e-07, "loss": 0.0, "num_tokens": 8670508.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1576 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03911695398734962, "grad_norm": 0.0, "learning_rate": 2.12e-07, "loss": 0.0, "num_tokens": 8671338.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7268.0, "completions/mean_length": 7730.0, "completions/mean_terminated_length": 7268.0, "completions/min_length": 7268.0, "completions/min_terminated_length": 7268.0, "epoch": 0.039141758650626315, "grad_norm": 3.193849563598633, "learning_rate": 2.1149999999999998e-07, "loss": -0.707, "num_tokens": 8679530.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1010.0, "completions/max_terminated_length": 1010.0, "completions/mean_length": 773.5, "completions/mean_terminated_length": 773.5, "completions/min_length": 537.0, "completions/min_terminated_length": 537.0, "epoch": 0.039166563313903016, "grad_norm": 0.0, "learning_rate": 2.1099999999999997e-07, "loss": 0.0, "num_tokens": 8681975.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1579 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3094.0, "completions/max_terminated_length": 3094.0, "completions/mean_length": 2767.0, "completions/mean_terminated_length": 2767.0, "completions/min_length": 2440.0, "completions/min_terminated_length": 2440.0, "epoch": 0.03919136797717971, "grad_norm": 0.0, "learning_rate": 2.1049999999999999e-07, "loss": 0.0, "num_tokens": 8688395.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4357.0, "completions/mean_length": 6274.5, "completions/mean_terminated_length": 4357.0, "completions/min_length": 4357.0, "completions/min_terminated_length": 4357.0, "epoch": 0.039216172640456404, "grad_norm": 4.618129730224609, "learning_rate": 2.0999999999999997e-07, "loss": -0.707, "num_tokens": 8693624.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6273.0, "completions/mean_length": 7232.5, "completions/mean_terminated_length": 6273.0, "completions/min_length": 6273.0, "completions/min_terminated_length": 6273.0, "epoch": 0.039240977303733104, "grad_norm": 2.9621164798736572, "learning_rate": 2.095e-07, "loss": -0.707, "num_tokens": 8700715.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1582 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0392657819670098, "grad_norm": 0.0, "learning_rate": 2.0899999999999998e-07, "loss": 0.0, "num_tokens": 8701581.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1583 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5630.0, "completions/max_terminated_length": 5630.0, "completions/mean_length": 4799.5, "completions/mean_terminated_length": 4799.5, "completions/min_length": 3969.0, "completions/min_terminated_length": 3969.0, "epoch": 0.03929058663028649, "grad_norm": 0.0, "learning_rate": 2.085e-07, "loss": 0.0, "num_tokens": 8712168.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1584 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03931539129356319, "grad_norm": 0.0, "learning_rate": 2.0799999999999998e-07, "loss": 0.0, "num_tokens": 8713280.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5343.0, "completions/max_terminated_length": 5343.0, "completions/mean_length": 5163.5, "completions/mean_terminated_length": 5163.5, "completions/min_length": 4984.0, "completions/min_terminated_length": 4984.0, "epoch": 0.039340195956839887, "grad_norm": 0.0, "learning_rate": 2.0749999999999997e-07, "loss": 0.0, "num_tokens": 8724503.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6897.0, "completions/max_terminated_length": 6897.0, "completions/mean_length": 6629.5, "completions/mean_terminated_length": 6629.5, "completions/min_length": 6362.0, "completions/min_terminated_length": 6362.0, "epoch": 0.03936500062011658, "grad_norm": 0.0, "learning_rate": 2.07e-07, "loss": 0.0, "num_tokens": 8738640.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4145.0, "completions/max_terminated_length": 4145.0, "completions/mean_length": 3264.0, "completions/mean_terminated_length": 3264.0, "completions/min_length": 2383.0, "completions/min_terminated_length": 2383.0, "epoch": 0.03938980528339328, "grad_norm": 0.0, "learning_rate": 2.0649999999999998e-07, "loss": 0.0, "num_tokens": 8745998.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8011.0, "completions/max_terminated_length": 8011.0, "completions/mean_length": 6859.5, "completions/mean_terminated_length": 6859.5, "completions/min_length": 5708.0, "completions/min_terminated_length": 5708.0, "epoch": 0.039414609946669975, "grad_norm": 0.0, "learning_rate": 2.06e-07, "loss": 0.0, "num_tokens": 8760597.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5811.0, "completions/mean_length": 7001.5, "completions/mean_terminated_length": 5811.0, "completions/min_length": 5811.0, "completions/min_terminated_length": 5811.0, "epoch": 0.03943941460994667, "grad_norm": 3.239457368850708, "learning_rate": 2.0549999999999998e-07, "loss": -0.707, "num_tokens": 8767288.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1590 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03946421927322337, "grad_norm": 0.0, "learning_rate": 2.0499999999999997e-07, "loss": 0.0, "num_tokens": 8768222.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1591 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2886.0, "completions/max_terminated_length": 2886.0, "completions/mean_length": 2667.0, "completions/mean_terminated_length": 2667.0, "completions/min_length": 2448.0, "completions/min_terminated_length": 2448.0, "epoch": 0.03948902393650006, "grad_norm": 0.0, "learning_rate": 2.0449999999999998e-07, "loss": 0.0, "num_tokens": 8774416.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1592 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03951382859977676, "grad_norm": 0.0, "learning_rate": 2.0399999999999997e-07, "loss": 0.0, "num_tokens": 8775312.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3797.0, "completions/mean_length": 5994.5, "completions/mean_terminated_length": 3797.0, "completions/min_length": 3797.0, "completions/min_terminated_length": 3797.0, "epoch": 0.03953863326305345, "grad_norm": 0.0, "learning_rate": 2.035e-07, "loss": 0.0, "num_tokens": 8780065.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2086.0, "completions/max_terminated_length": 2086.0, "completions/mean_length": 1599.5, "completions/mean_terminated_length": 1599.5, "completions/min_length": 1113.0, "completions/min_terminated_length": 1113.0, "epoch": 0.03956343792633015, "grad_norm": 0.0, "learning_rate": 2.03e-07, "loss": 0.0, "num_tokens": 8784116.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7507.0, "completions/max_terminated_length": 7507.0, "completions/mean_length": 7089.5, "completions/mean_terminated_length": 7089.5, "completions/min_length": 6672.0, "completions/min_terminated_length": 6672.0, "epoch": 0.039588242589606845, "grad_norm": 2.7314867973327637, "learning_rate": 2.025e-07, "loss": 0.0416, "num_tokens": 8799197.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 630.0, "completions/max_terminated_length": 630.0, "completions/mean_length": 558.0, "completions/mean_terminated_length": 558.0, "completions/min_length": 486.0, "completions/min_terminated_length": 486.0, "epoch": 0.03961304725288354, "grad_norm": 0.0, "learning_rate": 2.02e-07, "loss": 0.0, "num_tokens": 8801121.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4760.0, "completions/mean_length": 6476.0, "completions/mean_terminated_length": 4760.0, "completions/min_length": 4760.0, "completions/min_terminated_length": 4760.0, "epoch": 0.03963785191616024, "grad_norm": 5.592315196990967, "learning_rate": 2.015e-07, "loss": -0.707, "num_tokens": 8806873.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5922.0, "completions/max_terminated_length": 5922.0, "completions/mean_length": 4258.0, "completions/mean_terminated_length": 4258.0, "completions/min_length": 2594.0, "completions/min_terminated_length": 2594.0, "epoch": 0.03966265657943693, "grad_norm": 0.0, "learning_rate": 2.01e-07, "loss": 0.0, "num_tokens": 8816283.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1599 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03968746124271363, "grad_norm": 0.0, "learning_rate": 2.005e-07, "loss": 0.0, "num_tokens": 8817141.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7372.0, "completions/mean_length": 7782.0, "completions/mean_terminated_length": 7372.0, "completions/min_length": 7372.0, "completions/min_terminated_length": 7372.0, "epoch": 0.03971226590599033, "grad_norm": 3.273000717163086, "learning_rate": 2e-07, "loss": -0.707, "num_tokens": 8825529.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1601 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.03973707056926702, "grad_norm": 0.0, "learning_rate": 1.995e-07, "loss": 0.0, "num_tokens": 8826669.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1602 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.039761875232543716, "grad_norm": 0.0, "learning_rate": 1.99e-07, "loss": 0.0, "num_tokens": 8827613.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1603 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.039786679895820416, "grad_norm": 0.0, "learning_rate": 1.985e-07, "loss": 0.0, "num_tokens": 8828511.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1958.0, "completions/max_terminated_length": 1958.0, "completions/mean_length": 1739.5, "completions/mean_terminated_length": 1739.5, "completions/min_length": 1521.0, "completions/min_terminated_length": 1521.0, "epoch": 0.03981148455909711, "grad_norm": 0.0, "learning_rate": 1.98e-07, "loss": 0.0, "num_tokens": 8833090.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1605 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.039836289222373804, "grad_norm": 0.0, "learning_rate": 1.975e-07, "loss": 0.0, "num_tokens": 8833984.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1762.0, "completions/mean_length": 4977.0, "completions/mean_terminated_length": 1762.0, "completions/min_length": 1762.0, "completions/min_terminated_length": 1762.0, "epoch": 0.039861093885650505, "grad_norm": 0.0, "learning_rate": 1.97e-07, "loss": 0.0, "num_tokens": 8836726.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1607 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7221.0, "completions/max_terminated_length": 7221.0, "completions/mean_length": 4835.5, "completions/mean_terminated_length": 4835.5, "completions/min_length": 2450.0, "completions/min_terminated_length": 2450.0, "epoch": 0.0398858985489272, "grad_norm": 0.0, "learning_rate": 1.965e-07, "loss": 0.0, "num_tokens": 8847275.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 771.0, "completions/max_terminated_length": 771.0, "completions/mean_length": 744.0, "completions/mean_terminated_length": 744.0, "completions/min_length": 717.0, "completions/min_terminated_length": 717.0, "epoch": 0.03991070321220389, "grad_norm": 7.691936492919922, "learning_rate": 1.96e-07, "loss": -0.0257, "num_tokens": 8849555.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4386.0, "completions/max_terminated_length": 4386.0, "completions/mean_length": 2698.0, "completions/mean_terminated_length": 2698.0, "completions/min_length": 1010.0, "completions/min_terminated_length": 1010.0, "epoch": 0.03993550787548059, "grad_norm": 0.0, "learning_rate": 1.955e-07, "loss": 0.0, "num_tokens": 8856491.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1610 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4989.0, "completions/max_terminated_length": 4989.0, "completions/mean_length": 4133.5, "completions/mean_terminated_length": 4133.5, "completions/min_length": 3278.0, "completions/min_terminated_length": 3278.0, "epoch": 0.03996031253875729, "grad_norm": 3.0382442474365234, "learning_rate": 1.9499999999999999e-07, "loss": -0.1463, "num_tokens": 8865746.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1611 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6388.0, "completions/max_terminated_length": 6388.0, "completions/mean_length": 5181.5, "completions/mean_terminated_length": 5181.5, "completions/min_length": 3975.0, "completions/min_terminated_length": 3975.0, "epoch": 0.03998511720203398, "grad_norm": 2.960780620574951, "learning_rate": 1.945e-07, "loss": 0.1646, "num_tokens": 8876973.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1367.0, "completions/max_terminated_length": 1367.0, "completions/mean_length": 1166.0, "completions/mean_terminated_length": 1166.0, "completions/min_length": 965.0, "completions/min_terminated_length": 965.0, "epoch": 0.04000992186531068, "grad_norm": 4.959284782409668, "learning_rate": 1.94e-07, "loss": 0.1219, "num_tokens": 8880195.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1613 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7136.0, "completions/mean_length": 7664.0, "completions/mean_terminated_length": 7136.0, "completions/min_length": 7136.0, "completions/min_terminated_length": 7136.0, "epoch": 0.040034726528587375, "grad_norm": 3.6390597820281982, "learning_rate": 1.935e-07, "loss": -0.707, "num_tokens": 8888215.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6393.0, "completions/mean_length": 7292.5, "completions/mean_terminated_length": 6393.0, "completions/min_length": 6393.0, "completions/min_terminated_length": 6393.0, "epoch": 0.04005953119186407, "grad_norm": 0.0, "learning_rate": 1.93e-07, "loss": 0.0, "num_tokens": 8895620.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4334.0, "completions/max_terminated_length": 4334.0, "completions/mean_length": 4008.0, "completions/mean_terminated_length": 4008.0, "completions/min_length": 3682.0, "completions/min_terminated_length": 3682.0, "epoch": 0.04008433585514077, "grad_norm": 0.0, "learning_rate": 1.9249999999999998e-07, "loss": 0.0, "num_tokens": 8905684.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6711.0, "completions/mean_length": 7451.5, "completions/mean_terminated_length": 6711.0, "completions/min_length": 6711.0, "completions/min_terminated_length": 6711.0, "epoch": 0.04010914051841746, "grad_norm": 2.8045949935913086, "learning_rate": 1.92e-07, "loss": -0.707, "num_tokens": 8913323.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1617 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04013394518169416, "grad_norm": 0.0, "learning_rate": 1.915e-07, "loss": 0.0, "num_tokens": 8914301.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1618 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04015874984497086, "grad_norm": 0.0, "learning_rate": 1.91e-07, "loss": 0.0, "num_tokens": 8915361.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1619 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 556.0, "completions/max_terminated_length": 556.0, "completions/mean_length": 541.0, "completions/mean_terminated_length": 541.0, "completions/min_length": 526.0, "completions/min_terminated_length": 526.0, "epoch": 0.04018355450824755, "grad_norm": 0.0, "learning_rate": 1.905e-07, "loss": 0.0, "num_tokens": 8917239.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1620 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 883.0, "completions/max_terminated_length": 883.0, "completions/mean_length": 738.0, "completions/mean_terminated_length": 738.0, "completions/min_length": 593.0, "completions/min_terminated_length": 593.0, "epoch": 0.040208359171524245, "grad_norm": 0.0, "learning_rate": 1.8999999999999998e-07, "loss": 0.0, "num_tokens": 8919645.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1621 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3928.0, "completions/max_terminated_length": 3928.0, "completions/mean_length": 3649.0, "completions/mean_terminated_length": 3649.0, "completions/min_length": 3370.0, "completions/min_terminated_length": 3370.0, "epoch": 0.040233163834800946, "grad_norm": 2.6676225662231445, "learning_rate": 1.895e-07, "loss": 0.0541, "num_tokens": 8927901.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1622 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3660.0, "completions/max_terminated_length": 3660.0, "completions/mean_length": 3322.0, "completions/mean_terminated_length": 3322.0, "completions/min_length": 2984.0, "completions/min_terminated_length": 2984.0, "epoch": 0.04025796849807764, "grad_norm": 2.95961594581604, "learning_rate": 1.8899999999999999e-07, "loss": 0.0719, "num_tokens": 8935715.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1623 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.040282773161354334, "grad_norm": 0.0, "learning_rate": 1.885e-07, "loss": 0.0, "num_tokens": 8936581.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3319.0, "completions/max_terminated_length": 3319.0, "completions/mean_length": 2577.0, "completions/mean_terminated_length": 2577.0, "completions/min_length": 1835.0, "completions/min_terminated_length": 1835.0, "epoch": 0.04030757782463103, "grad_norm": 0.0, "learning_rate": 1.88e-07, "loss": 0.0, "num_tokens": 8942613.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4467.0, "completions/max_terminated_length": 4467.0, "completions/mean_length": 3475.5, "completions/mean_terminated_length": 3475.5, "completions/min_length": 2484.0, "completions/min_terminated_length": 2484.0, "epoch": 0.04033238248790773, "grad_norm": 0.0, "learning_rate": 1.875e-07, "loss": 0.0, "num_tokens": 8950420.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1626 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04035718715118442, "grad_norm": 0.0, "learning_rate": 1.87e-07, "loss": 0.0, "num_tokens": 8951412.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1926.0, "completions/max_terminated_length": 1926.0, "completions/mean_length": 1848.5, "completions/mean_terminated_length": 1848.5, "completions/min_length": 1771.0, "completions/min_terminated_length": 1771.0, "epoch": 0.040381991814461116, "grad_norm": 0.0, "learning_rate": 1.8649999999999998e-07, "loss": 0.0, "num_tokens": 8955985.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7370.0, "completions/max_terminated_length": 7370.0, "completions/mean_length": 6647.0, "completions/mean_terminated_length": 6647.0, "completions/min_length": 5924.0, "completions/min_terminated_length": 5924.0, "epoch": 0.040406796477737816, "grad_norm": 0.0, "learning_rate": 1.86e-07, "loss": 0.0, "num_tokens": 8970205.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6012.0, "completions/max_terminated_length": 6012.0, "completions/mean_length": 4494.5, "completions/mean_terminated_length": 4494.5, "completions/min_length": 2977.0, "completions/min_terminated_length": 2977.0, "epoch": 0.04043160114101451, "grad_norm": 4.090883731842041, "learning_rate": 1.855e-07, "loss": 0.2387, "num_tokens": 8981242.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1630 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8008.0, "completions/mean_length": 8100.0, "completions/mean_terminated_length": 8008.0, "completions/min_length": 8008.0, "completions/min_terminated_length": 8008.0, "epoch": 0.040456405804291204, "grad_norm": 0.0, "learning_rate": 1.85e-07, "loss": 0.0, "num_tokens": 8990098.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6268.0, "completions/mean_length": 7230.0, "completions/mean_terminated_length": 6268.0, "completions/min_length": 6268.0, "completions/min_terminated_length": 6268.0, "epoch": 0.040481210467567905, "grad_norm": 0.0, "learning_rate": 1.845e-07, "loss": 0.0, "num_tokens": 8997454.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1632 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4407.0, "completions/mean_length": 6299.5, "completions/mean_terminated_length": 4407.0, "completions/min_length": 4407.0, "completions/min_terminated_length": 4407.0, "epoch": 0.0405060151308446, "grad_norm": 3.8219594955444336, "learning_rate": 1.8399999999999998e-07, "loss": -0.707, "num_tokens": 9002913.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6171.0, "completions/max_terminated_length": 6171.0, "completions/mean_length": 4025.5, "completions/mean_terminated_length": 4025.5, "completions/min_length": 1880.0, "completions/min_terminated_length": 1880.0, "epoch": 0.04053081979412129, "grad_norm": 2.982229709625244, "learning_rate": 1.835e-07, "loss": 0.3768, "num_tokens": 9011766.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6175.0, "completions/mean_length": 7183.5, "completions/mean_terminated_length": 6175.0, "completions/min_length": 6175.0, "completions/min_terminated_length": 6175.0, "epoch": 0.04055562445739799, "grad_norm": 4.0082621574401855, "learning_rate": 1.8299999999999998e-07, "loss": -0.707, "num_tokens": 9018809.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3872.0, "completions/max_terminated_length": 3872.0, "completions/mean_length": 3841.5, "completions/mean_terminated_length": 3841.5, "completions/min_length": 3811.0, "completions/min_terminated_length": 3811.0, "epoch": 0.04058042912067469, "grad_norm": 0.0, "learning_rate": 1.825e-07, "loss": 0.0, "num_tokens": 9027350.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1163.0, "completions/max_terminated_length": 1163.0, "completions/mean_length": 814.5, "completions/mean_terminated_length": 814.5, "completions/min_length": 466.0, "completions/min_terminated_length": 466.0, "epoch": 0.04060523378395138, "grad_norm": 0.0, "learning_rate": 1.82e-07, "loss": 0.0, "num_tokens": 9029907.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1637 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8122.0, "completions/mean_length": 8157.0, "completions/mean_terminated_length": 8122.0, "completions/min_length": 8122.0, "completions/min_terminated_length": 8122.0, "epoch": 0.04063003844722808, "grad_norm": 0.0, "learning_rate": 1.8149999999999998e-07, "loss": 0.0, "num_tokens": 9038923.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3588.0, "completions/mean_length": 5890.0, "completions/mean_terminated_length": 3588.0, "completions/min_length": 3588.0, "completions/min_terminated_length": 3588.0, "epoch": 0.040654843110504775, "grad_norm": 3.8437979221343994, "learning_rate": 1.81e-07, "loss": -0.707, "num_tokens": 9043475.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1639 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04067964777378147, "grad_norm": 0.0, "learning_rate": 1.8049999999999998e-07, "loss": 0.0, "num_tokens": 9044403.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7615.0, "completions/mean_length": 7903.5, "completions/mean_terminated_length": 7615.0, "completions/min_length": 7615.0, "completions/min_terminated_length": 7615.0, "epoch": 0.04070445243705817, "grad_norm": 0.0, "learning_rate": 1.8e-07, "loss": 0.0, "num_tokens": 9052956.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1641 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5704.0, "completions/mean_length": 6948.0, "completions/mean_terminated_length": 5704.0, "completions/min_length": 5704.0, "completions/min_terminated_length": 5704.0, "epoch": 0.04072925710033486, "grad_norm": 0.0, "learning_rate": 1.7949999999999999e-07, "loss": 0.0, "num_tokens": 9059544.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1642 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1992.0, "completions/max_terminated_length": 1992.0, "completions/mean_length": 1983.0, "completions/mean_terminated_length": 1983.0, "completions/min_length": 1974.0, "completions/min_terminated_length": 1974.0, "epoch": 0.04075406176361156, "grad_norm": 0.0, "learning_rate": 1.7899999999999997e-07, "loss": 0.0, "num_tokens": 9064332.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6515.0, "completions/max_terminated_length": 6515.0, "completions/mean_length": 6114.0, "completions/mean_terminated_length": 6114.0, "completions/min_length": 5713.0, "completions/min_terminated_length": 5713.0, "epoch": 0.04077886642688826, "grad_norm": 0.0, "learning_rate": 1.785e-07, "loss": 0.0, "num_tokens": 9077380.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 782.0, "completions/max_terminated_length": 782.0, "completions/mean_length": 773.0, "completions/mean_terminated_length": 773.0, "completions/min_length": 764.0, "completions/min_terminated_length": 764.0, "epoch": 0.04080367109016495, "grad_norm": 0.0, "learning_rate": 1.7799999999999998e-07, "loss": 0.0, "num_tokens": 9079776.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6526.0, "completions/mean_length": 7359.0, "completions/mean_terminated_length": 6526.0, "completions/min_length": 6526.0, "completions/min_terminated_length": 6526.0, "epoch": 0.040828475753441645, "grad_norm": 3.4864370822906494, "learning_rate": 1.775e-07, "loss": -0.707, "num_tokens": 9087200.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5722.0, "completions/max_terminated_length": 5722.0, "completions/mean_length": 4105.0, "completions/mean_terminated_length": 4105.0, "completions/min_length": 2488.0, "completions/min_terminated_length": 2488.0, "epoch": 0.040853280416718346, "grad_norm": 0.0, "learning_rate": 1.7699999999999998e-07, "loss": 0.0, "num_tokens": 9096288.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2128.0, "completions/max_terminated_length": 2128.0, "completions/mean_length": 1999.5, "completions/mean_terminated_length": 1999.5, "completions/min_length": 1871.0, "completions/min_terminated_length": 1871.0, "epoch": 0.04087808507999504, "grad_norm": 0.0, "learning_rate": 1.7649999999999997e-07, "loss": 0.0, "num_tokens": 9101129.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1648 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.040902889743271734, "grad_norm": 0.0, "learning_rate": 1.76e-07, "loss": 0.0, "num_tokens": 9102355.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2547.0, "completions/max_terminated_length": 2547.0, "completions/mean_length": 2388.5, "completions/mean_terminated_length": 2388.5, "completions/min_length": 2230.0, "completions/min_terminated_length": 2230.0, "epoch": 0.040927694406548434, "grad_norm": 0.0, "learning_rate": 1.7549999999999998e-07, "loss": 0.0, "num_tokens": 9108010.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1650 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1757.0, "completions/max_terminated_length": 1757.0, "completions/mean_length": 1329.5, "completions/mean_terminated_length": 1329.5, "completions/min_length": 902.0, "completions/min_terminated_length": 902.0, "epoch": 0.04095249906982513, "grad_norm": 0.0, "learning_rate": 1.75e-07, "loss": 0.0, "num_tokens": 9111523.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1651 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1875.0, "completions/max_terminated_length": 1875.0, "completions/mean_length": 1762.0, "completions/mean_terminated_length": 1762.0, "completions/min_length": 1649.0, "completions/min_terminated_length": 1649.0, "epoch": 0.04097730373310182, "grad_norm": 0.0, "learning_rate": 1.7449999999999998e-07, "loss": 0.0, "num_tokens": 9115929.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1652 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5376.0, "completions/max_terminated_length": 5376.0, "completions/mean_length": 4524.5, "completions/mean_terminated_length": 4524.5, "completions/min_length": 3673.0, "completions/min_terminated_length": 3673.0, "epoch": 0.041002108396378516, "grad_norm": 0.0, "learning_rate": 1.7399999999999997e-07, "loss": 0.0, "num_tokens": 9125912.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1653 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.041026913059655216, "grad_norm": 0.0, "learning_rate": 1.7349999999999999e-07, "loss": 0.0, "num_tokens": 9126770.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1654 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04105171772293191, "grad_norm": 0.0, "learning_rate": 1.7299999999999997e-07, "loss": 0.0, "num_tokens": 9127640.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2673.0, "completions/max_terminated_length": 2673.0, "completions/mean_length": 2218.5, "completions/mean_terminated_length": 2218.5, "completions/min_length": 1764.0, "completions/min_terminated_length": 1764.0, "epoch": 0.041076522386208604, "grad_norm": 0.0, "learning_rate": 1.725e-07, "loss": 0.0, "num_tokens": 9132909.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 861.0, "completions/mean_length": 4526.5, "completions/mean_terminated_length": 861.0, "completions/min_length": 861.0, "completions/min_terminated_length": 861.0, "epoch": 0.041101327049485305, "grad_norm": 9.928726196289062, "learning_rate": 1.7199999999999998e-07, "loss": -0.707, "num_tokens": 9134628.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1657 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3584.0, "completions/max_terminated_length": 3584.0, "completions/mean_length": 3247.0, "completions/mean_terminated_length": 3247.0, "completions/min_length": 2910.0, "completions/min_terminated_length": 2910.0, "epoch": 0.041126131712762, "grad_norm": 0.0, "learning_rate": 1.715e-07, "loss": 0.0, "num_tokens": 9142094.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3675.0, "completions/max_terminated_length": 3675.0, "completions/mean_length": 3535.0, "completions/mean_terminated_length": 3535.0, "completions/min_length": 3395.0, "completions/min_terminated_length": 3395.0, "epoch": 0.04115093637603869, "grad_norm": 0.0, "learning_rate": 1.71e-07, "loss": 0.0, "num_tokens": 9149992.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1659 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3720.0, "completions/max_terminated_length": 3720.0, "completions/mean_length": 2383.0, "completions/mean_terminated_length": 2383.0, "completions/min_length": 1046.0, "completions/min_terminated_length": 1046.0, "epoch": 0.04117574103931539, "grad_norm": 0.0, "learning_rate": 1.705e-07, "loss": 0.0, "num_tokens": 9155574.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1660 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1642.0, "completions/max_terminated_length": 1642.0, "completions/mean_length": 1312.5, "completions/mean_terminated_length": 1312.5, "completions/min_length": 983.0, "completions/min_terminated_length": 983.0, "epoch": 0.04120054570259209, "grad_norm": 0.0, "learning_rate": 1.7000000000000001e-07, "loss": 0.0, "num_tokens": 9159091.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7619.0, "completions/mean_length": 7905.5, "completions/mean_terminated_length": 7619.0, "completions/min_length": 7619.0, "completions/min_terminated_length": 7619.0, "epoch": 0.04122535036586878, "grad_norm": 0.0, "learning_rate": 1.695e-07, "loss": 0.0, "num_tokens": 9167720.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7992.0, "completions/mean_length": 8092.0, "completions/mean_terminated_length": 7992.0, "completions/min_length": 7992.0, "completions/min_terminated_length": 7992.0, "epoch": 0.04125015502914548, "grad_norm": 0.0, "learning_rate": 1.69e-07, "loss": 0.0, "num_tokens": 9176614.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1663 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.041274959692422175, "grad_norm": 0.0, "learning_rate": 1.685e-07, "loss": 0.0, "num_tokens": 9177698.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6509.0, "completions/max_terminated_length": 6509.0, "completions/mean_length": 5706.5, "completions/mean_terminated_length": 5706.5, "completions/min_length": 4904.0, "completions/min_terminated_length": 4904.0, "epoch": 0.04129976435569887, "grad_norm": 0.0, "learning_rate": 1.68e-07, "loss": 0.0, "num_tokens": 9190035.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6781.0, "completions/max_terminated_length": 6781.0, "completions/mean_length": 6014.5, "completions/mean_terminated_length": 6014.5, "completions/min_length": 5248.0, "completions/min_terminated_length": 5248.0, "epoch": 0.04132456901897557, "grad_norm": 0.0, "learning_rate": 1.675e-07, "loss": 0.0, "num_tokens": 9202968.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4322.0, "completions/mean_length": 6257.0, "completions/mean_terminated_length": 4322.0, "completions/min_length": 4322.0, "completions/min_terminated_length": 4322.0, "epoch": 0.04134937368225226, "grad_norm": 4.480203628540039, "learning_rate": 1.67e-07, "loss": -0.707, "num_tokens": 9208138.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3660.0, "completions/max_terminated_length": 3660.0, "completions/mean_length": 2398.5, "completions/mean_terminated_length": 2398.5, "completions/min_length": 1137.0, "completions/min_terminated_length": 1137.0, "epoch": 0.04137417834552896, "grad_norm": 0.0, "learning_rate": 1.665e-07, "loss": 0.0, "num_tokens": 9213837.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1668 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4450.0, "completions/mean_length": 6321.0, "completions/mean_terminated_length": 4450.0, "completions/min_length": 4450.0, "completions/min_terminated_length": 4450.0, "epoch": 0.04139898300880566, "grad_norm": 4.136009216308594, "learning_rate": 1.66e-07, "loss": -0.707, "num_tokens": 9219161.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3193.0, "completions/max_terminated_length": 3193.0, "completions/mean_length": 3085.0, "completions/mean_terminated_length": 3085.0, "completions/min_length": 2977.0, "completions/min_terminated_length": 2977.0, "epoch": 0.04142378767208235, "grad_norm": 2.5833547115325928, "learning_rate": 1.655e-07, "loss": -0.0248, "num_tokens": 9226141.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1670 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 712.0, "completions/max_terminated_length": 712.0, "completions/mean_length": 652.5, "completions/mean_terminated_length": 652.5, "completions/min_length": 593.0, "completions/min_terminated_length": 593.0, "epoch": 0.041448592335359045, "grad_norm": 0.0, "learning_rate": 1.65e-07, "loss": 0.0, "num_tokens": 9228254.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1671 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3361.0, "completions/max_terminated_length": 3361.0, "completions/mean_length": 2949.0, "completions/mean_terminated_length": 2949.0, "completions/min_length": 2537.0, "completions/min_terminated_length": 2537.0, "epoch": 0.041473396998635746, "grad_norm": 3.798804759979248, "learning_rate": 1.645e-07, "loss": -0.0988, "num_tokens": 9235058.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1672 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04149820166191244, "grad_norm": 0.0, "learning_rate": 1.64e-07, "loss": 0.0, "num_tokens": 9236072.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1673 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8129.0, "completions/mean_length": 8160.5, "completions/mean_terminated_length": 8129.0, "completions/min_length": 8129.0, "completions/min_terminated_length": 8129.0, "epoch": 0.041523006325189134, "grad_norm": 0.0, "learning_rate": 1.635e-07, "loss": 0.0, "num_tokens": 9245115.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4604.0, "completions/mean_length": 6398.0, "completions/mean_terminated_length": 4604.0, "completions/min_length": 4604.0, "completions/min_terminated_length": 4604.0, "epoch": 0.041547810988465834, "grad_norm": 0.0, "learning_rate": 1.63e-07, "loss": 0.0, "num_tokens": 9250741.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4316.0, "completions/max_terminated_length": 4316.0, "completions/mean_length": 3744.5, "completions/mean_terminated_length": 3744.5, "completions/min_length": 3173.0, "completions/min_terminated_length": 3173.0, "epoch": 0.04157261565174253, "grad_norm": 3.530106782913208, "learning_rate": 1.625e-07, "loss": -0.1079, "num_tokens": 9259058.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1676 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04159742031501922, "grad_norm": 0.0, "learning_rate": 1.62e-07, "loss": 0.0, "num_tokens": 9260214.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1677 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04162222497829592, "grad_norm": 0.0, "learning_rate": 1.615e-07, "loss": 0.0, "num_tokens": 9261086.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1678 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8179.0, "completions/mean_length": 8185.5, "completions/mean_terminated_length": 8179.0, "completions/min_length": 8179.0, "completions/min_terminated_length": 8179.0, "epoch": 0.04164702964157262, "grad_norm": 2.9187614917755127, "learning_rate": 1.61e-07, "loss": -0.707, "num_tokens": 9270225.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1679 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04167183430484931, "grad_norm": 0.0, "learning_rate": 1.605e-07, "loss": 0.0, "num_tokens": 9271177.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1680 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6373.0, "completions/mean_length": 7282.5, "completions/mean_terminated_length": 6373.0, "completions/min_length": 6373.0, "completions/min_terminated_length": 6373.0, "epoch": 0.04169663896812601, "grad_norm": 0.0, "learning_rate": 1.6e-07, "loss": 0.0, "num_tokens": 9278500.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1681 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6463.0, "completions/max_terminated_length": 6463.0, "completions/mean_length": 5967.5, "completions/mean_terminated_length": 5967.5, "completions/min_length": 5472.0, "completions/min_terminated_length": 5472.0, "epoch": 0.041721443631402705, "grad_norm": 0.0, "learning_rate": 1.595e-07, "loss": 0.0, "num_tokens": 9291427.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1682 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0417462482946794, "grad_norm": 0.0, "learning_rate": 1.59e-07, "loss": 0.0, "num_tokens": 9292357.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1683 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3975.0, "completions/max_terminated_length": 3975.0, "completions/mean_length": 3669.0, "completions/mean_terminated_length": 3669.0, "completions/min_length": 3363.0, "completions/min_terminated_length": 3363.0, "epoch": 0.04177105295795609, "grad_norm": 0.0, "learning_rate": 1.585e-07, "loss": 0.0, "num_tokens": 9300545.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7976.0, "completions/max_terminated_length": 7976.0, "completions/mean_length": 5605.5, "completions/mean_terminated_length": 5605.5, "completions/min_length": 3235.0, "completions/min_terminated_length": 3235.0, "epoch": 0.04179585762123279, "grad_norm": 0.0, "learning_rate": 1.5799999999999999e-07, "loss": 0.0, "num_tokens": 9312726.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4701.0, "completions/max_terminated_length": 4701.0, "completions/mean_length": 4372.0, "completions/mean_terminated_length": 4372.0, "completions/min_length": 4043.0, "completions/min_terminated_length": 4043.0, "epoch": 0.04182066228450949, "grad_norm": 0.0, "learning_rate": 1.575e-07, "loss": 0.0, "num_tokens": 9322420.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1686 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7137.0, "completions/mean_length": 7664.5, "completions/mean_terminated_length": 7137.0, "completions/min_length": 7137.0, "completions/min_terminated_length": 7137.0, "epoch": 0.04184546694778618, "grad_norm": 0.0, "learning_rate": 1.57e-07, "loss": 0.0, "num_tokens": 9330433.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1687 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04187027161106288, "grad_norm": 0.0, "learning_rate": 1.565e-07, "loss": 0.0, "num_tokens": 9331443.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1688 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3171.0, "completions/max_terminated_length": 3171.0, "completions/mean_length": 2153.0, "completions/mean_terminated_length": 2153.0, "completions/min_length": 1135.0, "completions/min_terminated_length": 1135.0, "epoch": 0.041895076274339575, "grad_norm": 0.0, "learning_rate": 1.56e-07, "loss": 0.0, "num_tokens": 9336625.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1689 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3864.0, "completions/max_terminated_length": 3864.0, "completions/mean_length": 2501.0, "completions/mean_terminated_length": 2501.0, "completions/min_length": 1138.0, "completions/min_terminated_length": 1138.0, "epoch": 0.04191988093761627, "grad_norm": 0.0, "learning_rate": 1.5549999999999998e-07, "loss": 0.0, "num_tokens": 9342527.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1690 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04194468560089297, "grad_norm": 0.0, "learning_rate": 1.55e-07, "loss": 0.0, "num_tokens": 9343397.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7739.0, "completions/max_terminated_length": 7739.0, "completions/mean_length": 6254.0, "completions/mean_terminated_length": 6254.0, "completions/min_length": 4769.0, "completions/min_terminated_length": 4769.0, "epoch": 0.041969490264169663, "grad_norm": 0.0, "learning_rate": 1.545e-07, "loss": 0.0, "num_tokens": 9356779.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1683.0, "completions/max_terminated_length": 1683.0, "completions/mean_length": 1607.0, "completions/mean_terminated_length": 1607.0, "completions/min_length": 1531.0, "completions/min_terminated_length": 1531.0, "epoch": 0.04199429492744636, "grad_norm": 0.0, "learning_rate": 1.54e-07, "loss": 0.0, "num_tokens": 9360887.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1693 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3854.0, "completions/mean_length": 6023.0, "completions/mean_terminated_length": 3854.0, "completions/min_length": 3854.0, "completions/min_terminated_length": 3854.0, "epoch": 0.04201909959072306, "grad_norm": 3.52095365524292, "learning_rate": 1.535e-07, "loss": -0.707, "num_tokens": 9365625.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2300.0, "completions/max_terminated_length": 2300.0, "completions/mean_length": 2195.5, "completions/mean_terminated_length": 2195.5, "completions/min_length": 2091.0, "completions/min_terminated_length": 2091.0, "epoch": 0.04204390425399975, "grad_norm": 0.0, "learning_rate": 1.5299999999999998e-07, "loss": 0.0, "num_tokens": 9370882.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1695 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.042068708917276446, "grad_norm": 0.0, "learning_rate": 1.525e-07, "loss": 0.0, "num_tokens": 9371740.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7100.0, "completions/mean_length": 7646.0, "completions/mean_terminated_length": 7100.0, "completions/min_length": 7100.0, "completions/min_terminated_length": 7100.0, "epoch": 0.042093513580553146, "grad_norm": 0.0, "learning_rate": 1.5199999999999998e-07, "loss": 0.0, "num_tokens": 9379708.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4396.0, "completions/max_terminated_length": 4396.0, "completions/mean_length": 4019.5, "completions/mean_terminated_length": 4019.5, "completions/min_length": 3643.0, "completions/min_terminated_length": 3643.0, "epoch": 0.04211831824382984, "grad_norm": 0.0, "learning_rate": 1.515e-07, "loss": 0.0, "num_tokens": 9388701.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1698 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.042143122907106534, "grad_norm": 0.0, "learning_rate": 1.51e-07, "loss": 0.0, "num_tokens": 9389773.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2661.0, "completions/max_terminated_length": 2661.0, "completions/mean_length": 2157.5, "completions/mean_terminated_length": 2157.5, "completions/min_length": 1654.0, "completions/min_terminated_length": 1654.0, "epoch": 0.042167927570383235, "grad_norm": 3.6607460975646973, "learning_rate": 1.5049999999999998e-07, "loss": 0.165, "num_tokens": 9394930.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 8004.0, "completions/mean_length": 8098.0, "completions/mean_terminated_length": 8004.0, "completions/min_length": 8004.0, "completions/min_terminated_length": 8004.0, "epoch": 0.04219273223365993, "grad_norm": 0.0, "learning_rate": 1.5e-07, "loss": 0.0, "num_tokens": 9403812.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1701 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04221753689693662, "grad_norm": 0.0, "learning_rate": 1.4949999999999998e-07, "loss": 0.0, "num_tokens": 9404738.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3611.0, "completions/mean_length": 5901.5, "completions/mean_terminated_length": 3611.0, "completions/min_length": 3611.0, "completions/min_terminated_length": 3611.0, "epoch": 0.04224234156021332, "grad_norm": 0.0, "learning_rate": 1.49e-07, "loss": 0.0, "num_tokens": 9409349.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2026.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1511.5, "completions/mean_terminated_length": 1511.5, "completions/min_length": 997.0, "completions/min_terminated_length": 997.0, "epoch": 0.04226714622349002, "grad_norm": 0.0, "learning_rate": 1.4849999999999999e-07, "loss": 0.0, "num_tokens": 9413202.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5894.0, "completions/max_terminated_length": 5894.0, "completions/mean_length": 3734.0, "completions/mean_terminated_length": 3734.0, "completions/min_length": 1574.0, "completions/min_terminated_length": 1574.0, "epoch": 0.04229195088676671, "grad_norm": 3.3028342723846436, "learning_rate": 1.4799999999999998e-07, "loss": -0.409, "num_tokens": 9421534.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5751.0, "completions/max_terminated_length": 5751.0, "completions/mean_length": 4520.5, "completions/mean_terminated_length": 4520.5, "completions/min_length": 3290.0, "completions/min_terminated_length": 3290.0, "epoch": 0.04231675555004341, "grad_norm": 2.7087604999542236, "learning_rate": 1.475e-07, "loss": 0.1925, "num_tokens": 9431367.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4874.0, "completions/mean_length": 6533.0, "completions/mean_terminated_length": 4874.0, "completions/min_length": 4874.0, "completions/min_terminated_length": 4874.0, "epoch": 0.042341560213320105, "grad_norm": 3.6603469848632812, "learning_rate": 1.4699999999999998e-07, "loss": -0.707, "num_tokens": 9437081.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3300.0, "completions/mean_length": 5746.0, "completions/mean_terminated_length": 3300.0, "completions/min_length": 3300.0, "completions/min_terminated_length": 3300.0, "epoch": 0.0423663648765968, "grad_norm": 0.0, "learning_rate": 1.465e-07, "loss": 0.0, "num_tokens": 9441289.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1708 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2772.0, "completions/max_terminated_length": 2772.0, "completions/mean_length": 2365.0, "completions/mean_terminated_length": 2365.0, "completions/min_length": 1958.0, "completions/min_terminated_length": 1958.0, "epoch": 0.0423911695398735, "grad_norm": 0.0, "learning_rate": 1.4599999999999998e-07, "loss": 0.0, "num_tokens": 9446887.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1709 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2344.0, "completions/max_terminated_length": 2344.0, "completions/mean_length": 2054.5, "completions/mean_terminated_length": 2054.5, "completions/min_length": 1765.0, "completions/min_terminated_length": 1765.0, "epoch": 0.04241597420315019, "grad_norm": 3.8837077617645264, "learning_rate": 1.4549999999999997e-07, "loss": -0.0996, "num_tokens": 9451854.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6761.0, "completions/mean_length": 7476.5, "completions/mean_terminated_length": 6761.0, "completions/min_length": 6761.0, "completions/min_terminated_length": 6761.0, "epoch": 0.04244077886642689, "grad_norm": 0.0, "learning_rate": 1.45e-07, "loss": 0.0, "num_tokens": 9459773.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7602.0, "completions/mean_length": 7897.0, "completions/mean_terminated_length": 7602.0, "completions/min_length": 7602.0, "completions/min_terminated_length": 7602.0, "epoch": 0.04246558352970359, "grad_norm": 2.7740225791931152, "learning_rate": 1.4449999999999998e-07, "loss": -0.707, "num_tokens": 9468287.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3305.0, "completions/max_terminated_length": 3305.0, "completions/mean_length": 3100.0, "completions/mean_terminated_length": 3100.0, "completions/min_length": 2895.0, "completions/min_terminated_length": 2895.0, "epoch": 0.04249038819298028, "grad_norm": 3.1864891052246094, "learning_rate": 1.44e-07, "loss": -0.0468, "num_tokens": 9475375.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 400.5, "completions/mean_terminated_length": 400.5, "completions/min_length": 335.0, "completions/min_terminated_length": 335.0, "epoch": 0.042515192856256975, "grad_norm": 0.0, "learning_rate": 1.4349999999999998e-07, "loss": 0.0, "num_tokens": 9477044.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1714 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04253999751953367, "grad_norm": 0.0, "learning_rate": 1.4299999999999997e-07, "loss": 0.0, "num_tokens": 9477952.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7075.0, "completions/mean_length": 7633.5, "completions/mean_terminated_length": 7075.0, "completions/min_length": 7075.0, "completions/min_terminated_length": 7075.0, "epoch": 0.04256480218281037, "grad_norm": 0.0, "learning_rate": 1.4249999999999999e-07, "loss": 0.0, "num_tokens": 9485925.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8036.0, "completions/max_terminated_length": 8036.0, "completions/mean_length": 6375.5, "completions/mean_terminated_length": 6375.5, "completions/min_length": 4715.0, "completions/min_terminated_length": 4715.0, "epoch": 0.042589606846087064, "grad_norm": 2.4971046447753906, "learning_rate": 1.4199999999999997e-07, "loss": -0.1841, "num_tokens": 9499512.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5584.0, "completions/max_terminated_length": 5584.0, "completions/mean_length": 4886.0, "completions/mean_terminated_length": 4886.0, "completions/min_length": 4188.0, "completions/min_terminated_length": 4188.0, "epoch": 0.04261441150936376, "grad_norm": 0.0, "learning_rate": 1.415e-07, "loss": 0.0, "num_tokens": 9510208.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1718 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 791.0, "completions/max_terminated_length": 791.0, "completions/mean_length": 762.0, "completions/mean_terminated_length": 762.0, "completions/min_length": 733.0, "completions/min_terminated_length": 733.0, "epoch": 0.04263921617264046, "grad_norm": 0.0, "learning_rate": 1.4099999999999998e-07, "loss": 0.0, "num_tokens": 9512576.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1719 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04266402083591715, "grad_norm": 0.0, "learning_rate": 1.4050000000000002e-07, "loss": 0.0, "num_tokens": 9513524.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1720 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.042688825499193846, "grad_norm": 0.0, "learning_rate": 1.4e-07, "loss": 0.0, "num_tokens": 9514380.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7881.0, "completions/max_terminated_length": 7881.0, "completions/mean_length": 7537.0, "completions/mean_terminated_length": 7537.0, "completions/min_length": 7193.0, "completions/min_terminated_length": 7193.0, "epoch": 0.042713630162470546, "grad_norm": 2.3827919960021973, "learning_rate": 1.395e-07, "loss": -0.0323, "num_tokens": 9530306.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 536.0, "completions/max_terminated_length": 536.0, "completions/mean_length": 494.5, "completions/mean_terminated_length": 494.5, "completions/min_length": 453.0, "completions/min_terminated_length": 453.0, "epoch": 0.04273843482574724, "grad_norm": 0.0, "learning_rate": 1.3900000000000001e-07, "loss": 0.0, "num_tokens": 9532159.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1723 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.042763239489023934, "grad_norm": 0.0, "learning_rate": 1.385e-07, "loss": 0.0, "num_tokens": 9533241.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5889.0, "completions/max_terminated_length": 5889.0, "completions/mean_length": 4204.5, "completions/mean_terminated_length": 4204.5, "completions/min_length": 2520.0, "completions/min_terminated_length": 2520.0, "epoch": 0.042788044152300635, "grad_norm": 2.9928598403930664, "learning_rate": 1.3800000000000002e-07, "loss": -0.2833, "num_tokens": 9542628.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1725 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04281284881557733, "grad_norm": 0.0, "learning_rate": 1.375e-07, "loss": 0.0, "num_tokens": 9543504.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2475.0, "completions/max_terminated_length": 2475.0, "completions/mean_length": 2170.5, "completions/mean_terminated_length": 2170.5, "completions/min_length": 1866.0, "completions/min_terminated_length": 1866.0, "epoch": 0.04283765347885402, "grad_norm": 0.0, "learning_rate": 1.37e-07, "loss": 0.0, "num_tokens": 9548711.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3478.0, "completions/max_terminated_length": 3478.0, "completions/mean_length": 3036.5, "completions/mean_terminated_length": 3036.5, "completions/min_length": 2595.0, "completions/min_terminated_length": 2595.0, "epoch": 0.04286245814213072, "grad_norm": 0.0, "learning_rate": 1.365e-07, "loss": 0.0, "num_tokens": 9555594.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1728 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04288726280540742, "grad_norm": 0.0, "learning_rate": 1.36e-07, "loss": 0.0, "num_tokens": 9556446.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7384.0, "completions/max_terminated_length": 7384.0, "completions/mean_length": 6690.0, "completions/mean_terminated_length": 6690.0, "completions/min_length": 5996.0, "completions/min_terminated_length": 5996.0, "epoch": 0.04291206746868411, "grad_norm": 0.0, "learning_rate": 1.3550000000000002e-07, "loss": 0.0, "num_tokens": 9571050.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2190.0, "completions/max_terminated_length": 2190.0, "completions/mean_length": 1957.0, "completions/mean_terminated_length": 1957.0, "completions/min_length": 1724.0, "completions/min_terminated_length": 1724.0, "epoch": 0.04293687213196081, "grad_norm": 0.0, "learning_rate": 1.35e-07, "loss": 0.0, "num_tokens": 9575836.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4272.0, "completions/max_terminated_length": 4272.0, "completions/mean_length": 2809.5, "completions/mean_terminated_length": 2809.5, "completions/min_length": 1347.0, "completions/min_terminated_length": 1347.0, "epoch": 0.042961676795237505, "grad_norm": 3.7926218509674072, "learning_rate": 1.345e-07, "loss": 0.368, "num_tokens": 9582731.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1732 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 590.0, "completions/max_terminated_length": 590.0, "completions/mean_length": 489.5, "completions/mean_terminated_length": 489.5, "completions/min_length": 389.0, "completions/min_terminated_length": 389.0, "epoch": 0.0429864814585142, "grad_norm": 0.0, "learning_rate": 1.34e-07, "loss": 0.0, "num_tokens": 9584504.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1733 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1311.0, "completions/max_terminated_length": 1311.0, "completions/mean_length": 1236.0, "completions/mean_terminated_length": 1236.0, "completions/min_length": 1161.0, "completions/min_terminated_length": 1161.0, "epoch": 0.0430112861217909, "grad_norm": 0.0, "learning_rate": 1.335e-07, "loss": 0.0, "num_tokens": 9587808.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1824.0, "completions/max_terminated_length": 1824.0, "completions/mean_length": 1501.5, "completions/mean_terminated_length": 1501.5, "completions/min_length": 1179.0, "completions/min_terminated_length": 1179.0, "epoch": 0.04303609078506759, "grad_norm": 0.0, "learning_rate": 1.33e-07, "loss": 0.0, "num_tokens": 9591633.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1735 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04306089544834429, "grad_norm": 0.0, "learning_rate": 1.325e-07, "loss": 0.0, "num_tokens": 9592493.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2658.0, "completions/max_terminated_length": 2658.0, "completions/mean_length": 2183.0, "completions/mean_terminated_length": 2183.0, "completions/min_length": 1708.0, "completions/min_terminated_length": 1708.0, "epoch": 0.04308570011162099, "grad_norm": 0.0, "learning_rate": 1.32e-07, "loss": 0.0, "num_tokens": 9597711.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3709.0, "completions/max_terminated_length": 3709.0, "completions/mean_length": 3112.5, "completions/mean_terminated_length": 3112.5, "completions/min_length": 2516.0, "completions/min_terminated_length": 2516.0, "epoch": 0.04311050477489768, "grad_norm": 0.0, "learning_rate": 1.315e-07, "loss": 0.0, "num_tokens": 9604746.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6237.0, "completions/max_terminated_length": 6237.0, "completions/mean_length": 6111.0, "completions/mean_terminated_length": 6111.0, "completions/min_length": 5985.0, "completions/min_terminated_length": 5985.0, "epoch": 0.043135309438174375, "grad_norm": 0.0, "learning_rate": 1.31e-07, "loss": 0.0, "num_tokens": 9617838.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1739 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.043160114101451076, "grad_norm": 0.0, "learning_rate": 1.305e-07, "loss": 0.0, "num_tokens": 9619010.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1740 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4252.0, "completions/max_terminated_length": 4252.0, "completions/mean_length": 4055.5, "completions/mean_terminated_length": 4055.5, "completions/min_length": 3859.0, "completions/min_terminated_length": 3859.0, "epoch": 0.04318491876472777, "grad_norm": 0.0, "learning_rate": 1.3e-07, "loss": 0.0, "num_tokens": 9627971.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1741 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1865.0, "completions/max_terminated_length": 1865.0, "completions/mean_length": 1473.5, "completions/mean_terminated_length": 1473.5, "completions/min_length": 1082.0, "completions/min_terminated_length": 1082.0, "epoch": 0.043209723428004464, "grad_norm": 0.0, "learning_rate": 1.295e-07, "loss": 0.0, "num_tokens": 9631754.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1742 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7767.0, "completions/max_terminated_length": 7767.0, "completions/mean_length": 6875.0, "completions/mean_terminated_length": 6875.0, "completions/min_length": 5983.0, "completions/min_terminated_length": 5983.0, "epoch": 0.04323452809128116, "grad_norm": 2.5870554447174072, "learning_rate": 1.29e-07, "loss": 0.0917, "num_tokens": 9646372.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1743 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04325933275455786, "grad_norm": 0.0, "learning_rate": 1.285e-07, "loss": 0.0, "num_tokens": 9647234.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6481.0, "completions/mean_length": 7336.5, "completions/mean_terminated_length": 6481.0, "completions/min_length": 6481.0, "completions/min_terminated_length": 6481.0, "epoch": 0.04328413741783455, "grad_norm": 3.5646636486053467, "learning_rate": 1.28e-07, "loss": -0.707, "num_tokens": 9654637.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4010.0, "completions/max_terminated_length": 4010.0, "completions/mean_length": 3834.5, "completions/mean_terminated_length": 3834.5, "completions/min_length": 3659.0, "completions/min_terminated_length": 3659.0, "epoch": 0.043308942081111246, "grad_norm": 0.0, "learning_rate": 1.275e-07, "loss": 0.0, "num_tokens": 9663222.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1309.0, "completions/max_terminated_length": 1309.0, "completions/mean_length": 978.5, "completions/mean_terminated_length": 978.5, "completions/min_length": 648.0, "completions/min_terminated_length": 648.0, "epoch": 0.043333746744387946, "grad_norm": 0.0, "learning_rate": 1.2699999999999999e-07, "loss": 0.0, "num_tokens": 9666103.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1747 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04335855140766464, "grad_norm": 0.0, "learning_rate": 1.265e-07, "loss": 0.0, "num_tokens": 9666977.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1748 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7000.0, "completions/mean_length": 7596.0, "completions/mean_terminated_length": 7000.0, "completions/min_length": 7000.0, "completions/min_terminated_length": 7000.0, "epoch": 0.043383356070941334, "grad_norm": 0.0, "learning_rate": 1.26e-07, "loss": 0.0, "num_tokens": 9674931.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1749 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 4124.0, "completions/mean_terminated_length": 56.0, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.043408160734218035, "grad_norm": 0.0, "learning_rate": 1.255e-07, "loss": 0.0, "num_tokens": 9676029.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1750 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 3391.0, "completions/mean_length": 5791.5, "completions/mean_terminated_length": 3391.0, "completions/min_length": 3391.0, "completions/min_terminated_length": 3391.0, "epoch": 0.04343296539749473, "grad_norm": 0.0, "learning_rate": 1.25e-07, "loss": 0.0, "num_tokens": 9680320.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1751 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5296.0, "completions/mean_length": 6744.0, "completions/mean_terminated_length": 5296.0, "completions/min_length": 5296.0, "completions/min_terminated_length": 5296.0, "epoch": 0.04345777006077142, "grad_norm": 0.0, "learning_rate": 1.2449999999999998e-07, "loss": 0.0, "num_tokens": 9686766.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2323.0, "completions/max_terminated_length": 2323.0, "completions/mean_length": 2223.5, "completions/mean_terminated_length": 2223.5, "completions/min_length": 2124.0, "completions/min_terminated_length": 2124.0, "epoch": 0.04348257472404812, "grad_norm": 0.0, "learning_rate": 1.24e-07, "loss": 0.0, "num_tokens": 9692113.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1753 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1435.0, "completions/max_terminated_length": 1435.0, "completions/mean_length": 1350.5, "completions/mean_terminated_length": 1350.5, "completions/min_length": 1266.0, "completions/min_terminated_length": 1266.0, "epoch": 0.04350737938732482, "grad_norm": 0.0, "learning_rate": 1.235e-07, "loss": 0.0, "num_tokens": 9696098.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1754 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04353218405060151, "grad_norm": 0.0, "learning_rate": 1.23e-07, "loss": 0.0, "num_tokens": 9697032.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6951.0, "completions/mean_length": 7571.5, "completions/mean_terminated_length": 6951.0, "completions/min_length": 6951.0, "completions/min_terminated_length": 6951.0, "epoch": 0.04355698871387821, "grad_norm": 0.0, "learning_rate": 1.225e-07, "loss": 0.0, "num_tokens": 9704813.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3379.0, "completions/max_terminated_length": 3379.0, "completions/mean_length": 2572.5, "completions/mean_terminated_length": 2572.5, "completions/min_length": 1766.0, "completions/min_terminated_length": 1766.0, "epoch": 0.043581793377154905, "grad_norm": 0.0, "learning_rate": 1.2199999999999998e-07, "loss": 0.0, "num_tokens": 9710992.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4037.0, "completions/mean_length": 6114.5, "completions/mean_terminated_length": 4037.0, "completions/min_length": 4037.0, "completions/min_terminated_length": 4037.0, "epoch": 0.0436065980404316, "grad_norm": 2.9442365169525146, "learning_rate": 1.215e-07, "loss": -0.707, "num_tokens": 9716043.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1758 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0436314027037083, "grad_norm": 0.0, "learning_rate": 1.2099999999999998e-07, "loss": 0.0, "num_tokens": 9717043.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1759 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4892.0, "completions/max_terminated_length": 4892.0, "completions/mean_length": 4485.0, "completions/mean_terminated_length": 4485.0, "completions/min_length": 4078.0, "completions/min_terminated_length": 4078.0, "epoch": 0.04365620736698499, "grad_norm": 0.0, "learning_rate": 1.205e-07, "loss": 0.0, "num_tokens": 9726999.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1760 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1041.0, "completions/mean_length": 4616.5, "completions/mean_terminated_length": 1041.0, "completions/min_length": 1041.0, "completions/min_terminated_length": 1041.0, "epoch": 0.04368101203026169, "grad_norm": 0.0, "learning_rate": 1.2e-07, "loss": 0.0, "num_tokens": 9728926.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7094.0, "completions/max_terminated_length": 7094.0, "completions/mean_length": 5613.5, "completions/mean_terminated_length": 5613.5, "completions/min_length": 4133.0, "completions/min_terminated_length": 4133.0, "epoch": 0.04370581669353839, "grad_norm": 0.0, "learning_rate": 1.1949999999999998e-07, "loss": 0.0, "num_tokens": 9741015.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1762 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3134.0, "completions/max_terminated_length": 3134.0, "completions/mean_length": 2476.0, "completions/mean_terminated_length": 2476.0, "completions/min_length": 1818.0, "completions/min_terminated_length": 1818.0, "epoch": 0.04373062135681508, "grad_norm": 0.0, "learning_rate": 1.19e-07, "loss": 0.0, "num_tokens": 9746955.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3677.0, "completions/max_terminated_length": 3677.0, "completions/mean_length": 2702.5, "completions/mean_terminated_length": 2702.5, "completions/min_length": 1728.0, "completions/min_terminated_length": 1728.0, "epoch": 0.043755426020091776, "grad_norm": 0.0, "learning_rate": 1.1849999999999998e-07, "loss": 0.0, "num_tokens": 9753200.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 768.0, "completions/max_terminated_length": 768.0, "completions/mean_length": 756.5, "completions/mean_terminated_length": 756.5, "completions/min_length": 745.0, "completions/min_terminated_length": 745.0, "epoch": 0.043780230683368476, "grad_norm": 0.0, "learning_rate": 1.1799999999999998e-07, "loss": 0.0, "num_tokens": 9755807.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1765 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04380503534664517, "grad_norm": 0.0, "learning_rate": 1.1749999999999999e-07, "loss": 0.0, "num_tokens": 9756817.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7978.0, "completions/mean_length": 8085.0, "completions/mean_terminated_length": 7978.0, "completions/min_length": 7978.0, "completions/min_terminated_length": 7978.0, "epoch": 0.043829840009921864, "grad_norm": 0.0, "learning_rate": 1.17e-07, "loss": 0.0, "num_tokens": 9765713.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1767 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3487.0, "completions/max_terminated_length": 3487.0, "completions/mean_length": 2670.0, "completions/mean_terminated_length": 2670.0, "completions/min_length": 1853.0, "completions/min_terminated_length": 1853.0, "epoch": 0.043854644673198565, "grad_norm": 0.0, "learning_rate": 1.165e-07, "loss": 0.0, "num_tokens": 9771949.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1768 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04387944933647526, "grad_norm": 0.0, "learning_rate": 1.16e-07, "loss": 0.0, "num_tokens": 9772919.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1029.0, "completions/max_terminated_length": 1029.0, "completions/mean_length": 1016.0, "completions/mean_terminated_length": 1016.0, "completions/min_length": 1003.0, "completions/min_terminated_length": 1003.0, "epoch": 0.04390425399975195, "grad_norm": 0.0, "learning_rate": 1.155e-07, "loss": 0.0, "num_tokens": 9775757.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1770 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2603.0, "completions/max_terminated_length": 2603.0, "completions/mean_length": 2168.5, "completions/mean_terminated_length": 2168.5, "completions/min_length": 1734.0, "completions/min_terminated_length": 1734.0, "epoch": 0.04392905866302865, "grad_norm": 0.0, "learning_rate": 1.15e-07, "loss": 0.0, "num_tokens": 9781004.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2729.0, "completions/max_terminated_length": 2729.0, "completions/mean_length": 2170.5, "completions/mean_terminated_length": 2170.5, "completions/min_length": 1612.0, "completions/min_terminated_length": 1612.0, "epoch": 0.04395386332630535, "grad_norm": 0.0, "learning_rate": 1.145e-07, "loss": 0.0, "num_tokens": 9786161.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1772 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2886.0, "completions/max_terminated_length": 2886.0, "completions/mean_length": 2533.0, "completions/mean_terminated_length": 2533.0, "completions/min_length": 2180.0, "completions/min_terminated_length": 2180.0, "epoch": 0.04397866798958204, "grad_norm": 0.0, "learning_rate": 1.14e-07, "loss": 0.0, "num_tokens": 9792187.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1773 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 659.0, "completions/max_terminated_length": 659.0, "completions/mean_length": 645.0, "completions/mean_terminated_length": 645.0, "completions/min_length": 631.0, "completions/min_terminated_length": 631.0, "epoch": 0.044003472652858734, "grad_norm": 0.0, "learning_rate": 1.135e-07, "loss": 0.0, "num_tokens": 9794297.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2608.0, "completions/max_terminated_length": 2608.0, "completions/mean_length": 2595.0, "completions/mean_terminated_length": 2595.0, "completions/min_length": 2582.0, "completions/min_terminated_length": 2582.0, "epoch": 0.044028277316135435, "grad_norm": 0.0, "learning_rate": 1.1299999999999999e-07, "loss": 0.0, "num_tokens": 9800327.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4975.0, "completions/max_terminated_length": 4975.0, "completions/mean_length": 4884.0, "completions/mean_terminated_length": 4884.0, "completions/min_length": 4793.0, "completions/min_terminated_length": 4793.0, "epoch": 0.04405308197941213, "grad_norm": 0.0, "learning_rate": 1.125e-07, "loss": 0.0, "num_tokens": 9810991.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6880.0, "completions/max_terminated_length": 6880.0, "completions/mean_length": 4873.0, "completions/mean_terminated_length": 4873.0, "completions/min_length": 2866.0, "completions/min_terminated_length": 2866.0, "epoch": 0.04407788664268882, "grad_norm": 0.0, "learning_rate": 1.12e-07, "loss": 0.0, "num_tokens": 9821599.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1395.0, "completions/max_terminated_length": 1395.0, "completions/mean_length": 1317.0, "completions/mean_terminated_length": 1317.0, "completions/min_length": 1239.0, "completions/min_terminated_length": 1239.0, "epoch": 0.04410269130596552, "grad_norm": 0.0, "learning_rate": 1.115e-07, "loss": 0.0, "num_tokens": 9825081.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1778 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6086.0, "completions/max_terminated_length": 6086.0, "completions/mean_length": 5025.5, "completions/mean_terminated_length": 5025.5, "completions/min_length": 3965.0, "completions/min_terminated_length": 3965.0, "epoch": 0.04412749596924222, "grad_norm": 0.0, "learning_rate": 1.11e-07, "loss": 0.0, "num_tokens": 9835984.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6211.0, "completions/max_terminated_length": 6211.0, "completions/mean_length": 6171.0, "completions/mean_terminated_length": 6171.0, "completions/min_length": 6131.0, "completions/min_terminated_length": 6131.0, "epoch": 0.04415230063251891, "grad_norm": 2.9687914848327637, "learning_rate": 1.1049999999999999e-07, "loss": -0.0046, "num_tokens": 9849274.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4815.0, "completions/max_terminated_length": 4815.0, "completions/mean_length": 3774.0, "completions/mean_terminated_length": 3774.0, "completions/min_length": 2733.0, "completions/min_terminated_length": 2733.0, "epoch": 0.04417710529579561, "grad_norm": 0.0, "learning_rate": 1.0999999999999999e-07, "loss": 0.0, "num_tokens": 9857664.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1781 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6151.0, "completions/mean_length": 7171.5, "completions/mean_terminated_length": 6151.0, "completions/min_length": 6151.0, "completions/min_terminated_length": 6151.0, "epoch": 0.044201909959072305, "grad_norm": 3.6413633823394775, "learning_rate": 1.095e-07, "loss": -0.707, "num_tokens": 9864749.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1782 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6025.0, "completions/mean_length": 7108.5, "completions/mean_terminated_length": 6025.0, "completions/min_length": 6025.0, "completions/min_terminated_length": 6025.0, "epoch": 0.044226714622349, "grad_norm": 0.0, "learning_rate": 1.09e-07, "loss": 0.0, "num_tokens": 9871824.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6298.0, "completions/max_terminated_length": 6298.0, "completions/mean_length": 4350.0, "completions/mean_terminated_length": 4350.0, "completions/min_length": 2402.0, "completions/min_terminated_length": 2402.0, "epoch": 0.0442515192856257, "grad_norm": 2.763782024383545, "learning_rate": 1.085e-07, "loss": 0.3166, "num_tokens": 9881370.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2189.0, "completions/max_terminated_length": 2189.0, "completions/mean_length": 1660.0, "completions/mean_terminated_length": 1660.0, "completions/min_length": 1131.0, "completions/min_terminated_length": 1131.0, "epoch": 0.044276323948902394, "grad_norm": 0.0, "learning_rate": 1.0799999999999999e-07, "loss": 0.0, "num_tokens": 9885518.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4569.0, "completions/mean_length": 6380.5, "completions/mean_terminated_length": 4569.0, "completions/min_length": 4569.0, "completions/min_terminated_length": 4569.0, "epoch": 0.04430112861217909, "grad_norm": 3.6199090480804443, "learning_rate": 1.0749999999999999e-07, "loss": -0.707, "num_tokens": 9891037.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1063.0, "completions/max_terminated_length": 1063.0, "completions/mean_length": 1042.0, "completions/mean_terminated_length": 1042.0, "completions/min_length": 1021.0, "completions/min_terminated_length": 1021.0, "epoch": 0.04432593327545579, "grad_norm": 0.0, "learning_rate": 1.0699999999999999e-07, "loss": 0.0, "num_tokens": 9893981.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5816.0, "completions/mean_length": 7004.0, "completions/mean_terminated_length": 5816.0, "completions/min_length": 5816.0, "completions/min_terminated_length": 5816.0, "epoch": 0.04435073793873248, "grad_norm": 4.9034905433654785, "learning_rate": 1.065e-07, "loss": -0.707, "num_tokens": 9900627.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1788 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.044375542602009176, "grad_norm": 0.0, "learning_rate": 1.06e-07, "loss": 0.0, "num_tokens": 9901623.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6090.0, "completions/mean_length": 7141.0, "completions/mean_terminated_length": 6090.0, "completions/min_length": 6090.0, "completions/min_terminated_length": 6090.0, "epoch": 0.044400347265285876, "grad_norm": 0.0, "learning_rate": 1.0549999999999999e-07, "loss": 0.0, "num_tokens": 9908743.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1790 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6826.0, "completions/mean_length": 7509.0, "completions/mean_terminated_length": 6826.0, "completions/min_length": 6826.0, "completions/min_terminated_length": 6826.0, "epoch": 0.04442515192856257, "grad_norm": 0.0, "learning_rate": 1.0499999999999999e-07, "loss": 0.0, "num_tokens": 9916685.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6448.0, "completions/max_terminated_length": 6448.0, "completions/mean_length": 5309.0, "completions/mean_terminated_length": 5309.0, "completions/min_length": 4170.0, "completions/min_terminated_length": 4170.0, "epoch": 0.044449956591839264, "grad_norm": 0.0, "learning_rate": 1.0449999999999999e-07, "loss": 0.0, "num_tokens": 9928185.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1792 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5872.0, "completions/max_terminated_length": 5872.0, "completions/mean_length": 4948.5, "completions/mean_terminated_length": 4948.5, "completions/min_length": 4025.0, "completions/min_terminated_length": 4025.0, "epoch": 0.044474761255115965, "grad_norm": 0.0, "learning_rate": 1.0399999999999999e-07, "loss": 0.0, "num_tokens": 9939188.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2155.0, "completions/max_terminated_length": 2155.0, "completions/mean_length": 1668.0, "completions/mean_terminated_length": 1668.0, "completions/min_length": 1181.0, "completions/min_terminated_length": 1181.0, "epoch": 0.04449956591839266, "grad_norm": 0.0, "learning_rate": 1.035e-07, "loss": 0.0, "num_tokens": 9943358.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3118.0, "completions/max_terminated_length": 3118.0, "completions/mean_length": 2942.0, "completions/mean_terminated_length": 2942.0, "completions/min_length": 2766.0, "completions/min_terminated_length": 2766.0, "epoch": 0.04452437058166935, "grad_norm": 0.0, "learning_rate": 1.03e-07, "loss": 0.0, "num_tokens": 9950082.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5446.0, "completions/mean_length": 6819.0, "completions/mean_terminated_length": 5446.0, "completions/min_length": 5446.0, "completions/min_terminated_length": 5446.0, "epoch": 0.04454917524494605, "grad_norm": 0.0, "learning_rate": 1.0249999999999998e-07, "loss": 0.0, "num_tokens": 9956356.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2531.0, "completions/max_terminated_length": 2531.0, "completions/mean_length": 1935.0, "completions/mean_terminated_length": 1935.0, "completions/min_length": 1339.0, "completions/min_terminated_length": 1339.0, "epoch": 0.04457397990822275, "grad_norm": 3.6414794921875, "learning_rate": 1.0199999999999999e-07, "loss": 0.2178, "num_tokens": 9961106.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1797 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04459878457149944, "grad_norm": 0.0, "learning_rate": 1.015e-07, "loss": 0.0, "num_tokens": 9962108.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4657.0, "completions/mean_length": 6424.5, "completions/mean_terminated_length": 4657.0, "completions/min_length": 4657.0, "completions/min_terminated_length": 4657.0, "epoch": 0.04462358923477614, "grad_norm": 4.352249622344971, "learning_rate": 1.01e-07, "loss": -0.707, "num_tokens": 9967653.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1799 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.044648393898052835, "grad_norm": 0.0, "learning_rate": 1.005e-07, "loss": 0.0, "num_tokens": 9968737.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6675.0, "completions/max_terminated_length": 6675.0, "completions/mean_length": 4544.0, "completions/mean_terminated_length": 4544.0, "completions/min_length": 2413.0, "completions/min_terminated_length": 2413.0, "epoch": 0.04467319856132953, "grad_norm": 0.0, "learning_rate": 1e-07, "loss": 0.0, "num_tokens": 9978671.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8174.0, "completions/max_terminated_length": 8174.0, "completions/mean_length": 6139.5, "completions/mean_terminated_length": 6139.5, "completions/min_length": 4105.0, "completions/min_terminated_length": 4105.0, "epoch": 0.04469800322460622, "grad_norm": 3.0087273120880127, "learning_rate": 9.95e-08, "loss": 0.2343, "num_tokens": 9991792.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1802 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 950.0, "completions/max_terminated_length": 950.0, "completions/mean_length": 862.0, "completions/mean_terminated_length": 862.0, "completions/min_length": 774.0, "completions/min_terminated_length": 774.0, "epoch": 0.04472280788788292, "grad_norm": 0.0, "learning_rate": 9.9e-08, "loss": 0.0, "num_tokens": 9994364.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1803 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04474761255115962, "grad_norm": 0.0, "learning_rate": 9.85e-08, "loss": 0.0, "num_tokens": 9995320.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3525.0, "completions/max_terminated_length": 3525.0, "completions/mean_length": 2785.0, "completions/mean_terminated_length": 2785.0, "completions/min_length": 2045.0, "completions/min_terminated_length": 2045.0, "epoch": 0.04477241721443631, "grad_norm": 0.0, "learning_rate": 9.8e-08, "loss": 0.0, "num_tokens": 10001710.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7023.0, "completions/mean_length": 7607.5, "completions/mean_terminated_length": 7023.0, "completions/min_length": 7023.0, "completions/min_terminated_length": 7023.0, "epoch": 0.04479722187771301, "grad_norm": 0.0, "learning_rate": 9.749999999999999e-08, "loss": 0.0, "num_tokens": 10009619.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6498.0, "completions/max_terminated_length": 6498.0, "completions/mean_length": 3670.5, "completions/mean_terminated_length": 3670.5, "completions/min_length": 843.0, "completions/min_terminated_length": 843.0, "epoch": 0.044822026540989705, "grad_norm": 0.0, "learning_rate": 9.7e-08, "loss": 0.0, "num_tokens": 10018024.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1807 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0448468312042664, "grad_norm": 0.0, "learning_rate": 9.65e-08, "loss": 0.0, "num_tokens": 10018994.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1808 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0448716358675431, "grad_norm": 0.0, "learning_rate": 9.6e-08, "loss": 0.0, "num_tokens": 10019954.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5277.0, "completions/max_terminated_length": 5277.0, "completions/mean_length": 4219.5, "completions/mean_terminated_length": 4219.5, "completions/min_length": 3162.0, "completions/min_terminated_length": 3162.0, "epoch": 0.044896440530819794, "grad_norm": 0.0, "learning_rate": 9.55e-08, "loss": 0.0, "num_tokens": 10029227.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1810 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7145.0, "completions/max_terminated_length": 7145.0, "completions/mean_length": 6331.5, "completions/mean_terminated_length": 6331.5, "completions/min_length": 5518.0, "completions/min_terminated_length": 5518.0, "epoch": 0.04492124519409649, "grad_norm": 0.0, "learning_rate": 9.499999999999999e-08, "loss": 0.0, "num_tokens": 10042736.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3625.0, "completions/max_terminated_length": 3625.0, "completions/mean_length": 3353.0, "completions/mean_terminated_length": 3353.0, "completions/min_length": 3081.0, "completions/min_terminated_length": 3081.0, "epoch": 0.04494604985737319, "grad_norm": 0.0, "learning_rate": 9.449999999999999e-08, "loss": 0.0, "num_tokens": 10050332.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 683.0, "completions/max_terminated_length": 683.0, "completions/mean_length": 651.5, "completions/mean_terminated_length": 651.5, "completions/min_length": 620.0, "completions/min_terminated_length": 620.0, "epoch": 0.04497085452064988, "grad_norm": 0.0, "learning_rate": 9.4e-08, "loss": 0.0, "num_tokens": 10052427.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1813 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.044995659183926576, "grad_norm": 0.0, "learning_rate": 9.35e-08, "loss": 0.0, "num_tokens": 10053383.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7170.0, "completions/mean_length": 7681.0, "completions/mean_terminated_length": 7170.0, "completions/min_length": 7170.0, "completions/min_terminated_length": 7170.0, "epoch": 0.045020463847203276, "grad_norm": 3.547973155975342, "learning_rate": 9.3e-08, "loss": -0.707, "num_tokens": 10061551.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7865.0, "completions/mean_length": 8028.5, "completions/mean_terminated_length": 7865.0, "completions/min_length": 7865.0, "completions/min_terminated_length": 7865.0, "epoch": 0.04504526851047997, "grad_norm": 0.0, "learning_rate": 9.25e-08, "loss": 0.0, "num_tokens": 10070372.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1816 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2515.0, "completions/max_terminated_length": 2515.0, "completions/mean_length": 2078.5, "completions/mean_terminated_length": 2078.5, "completions/min_length": 1642.0, "completions/min_terminated_length": 1642.0, "epoch": 0.045070073173756664, "grad_norm": 0.0, "learning_rate": 9.199999999999999e-08, "loss": 0.0, "num_tokens": 10075441.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1817 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.045094877837033365, "grad_norm": 0.0, "learning_rate": 9.149999999999999e-08, "loss": 0.0, "num_tokens": 10076379.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1818 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3629.0, "completions/max_terminated_length": 3629.0, "completions/mean_length": 3471.0, "completions/mean_terminated_length": 3471.0, "completions/min_length": 3313.0, "completions/min_terminated_length": 3313.0, "epoch": 0.04511968250031006, "grad_norm": 0.0, "learning_rate": 9.1e-08, "loss": 0.0, "num_tokens": 10084205.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2079.0, "completions/mean_length": 5135.5, "completions/mean_terminated_length": 2079.0, "completions/min_length": 2079.0, "completions/min_terminated_length": 2079.0, "epoch": 0.04514448716358675, "grad_norm": 0.0, "learning_rate": 9.05e-08, "loss": 0.0, "num_tokens": 10087274.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3899.0, "completions/max_terminated_length": 3899.0, "completions/mean_length": 3377.0, "completions/mean_terminated_length": 3377.0, "completions/min_length": 2855.0, "completions/min_terminated_length": 2855.0, "epoch": 0.04516929182686345, "grad_norm": 3.5509884357452393, "learning_rate": 9e-08, "loss": 0.1093, "num_tokens": 10094996.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 876.0, "completions/max_terminated_length": 876.0, "completions/mean_length": 860.5, "completions/mean_terminated_length": 860.5, "completions/min_length": 845.0, "completions/min_terminated_length": 845.0, "epoch": 0.04519409649014015, "grad_norm": 0.0, "learning_rate": 8.949999999999999e-08, "loss": 0.0, "num_tokens": 10097523.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 561.0, "completions/max_terminated_length": 561.0, "completions/mean_length": 550.0, "completions/mean_terminated_length": 550.0, "completions/min_length": 539.0, "completions/min_terminated_length": 539.0, "epoch": 0.04521890115341684, "grad_norm": 0.0, "learning_rate": 8.899999999999999e-08, "loss": 0.0, "num_tokens": 10099465.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1823 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04524370581669354, "grad_norm": 0.0, "learning_rate": 8.849999999999999e-08, "loss": 0.0, "num_tokens": 10100357.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 645.0, "completions/max_terminated_length": 645.0, "completions/mean_length": 568.5, "completions/mean_terminated_length": 568.5, "completions/min_length": 492.0, "completions/min_terminated_length": 492.0, "epoch": 0.045268510479970235, "grad_norm": 0.0, "learning_rate": 8.8e-08, "loss": 0.0, "num_tokens": 10102314.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1825 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04529331514324693, "grad_norm": 0.0, "learning_rate": 8.75e-08, "loss": 0.0, "num_tokens": 10103222.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1826 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04531811980652363, "grad_norm": 0.0, "learning_rate": 8.699999999999998e-08, "loss": 0.0, "num_tokens": 10104262.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1827 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04534292446980032, "grad_norm": 0.0, "learning_rate": 8.649999999999999e-08, "loss": 0.0, "num_tokens": 10105136.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1874.0, "completions/max_terminated_length": 1874.0, "completions/mean_length": 1272.5, "completions/mean_terminated_length": 1272.5, "completions/min_length": 671.0, "completions/min_terminated_length": 671.0, "epoch": 0.04536772913307702, "grad_norm": 0.0, "learning_rate": 8.599999999999999e-08, "loss": 0.0, "num_tokens": 10108517.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1829 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04539253379635372, "grad_norm": 0.0, "learning_rate": 8.55e-08, "loss": 0.0, "num_tokens": 10109355.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1830 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5004.0, "completions/max_terminated_length": 5004.0, "completions/mean_length": 4393.5, "completions/mean_terminated_length": 4393.5, "completions/min_length": 3783.0, "completions/min_terminated_length": 3783.0, "epoch": 0.04541733845963041, "grad_norm": 0.0, "learning_rate": 8.500000000000001e-08, "loss": 0.0, "num_tokens": 10119022.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5390.0, "completions/max_terminated_length": 5390.0, "completions/mean_length": 4510.0, "completions/mean_terminated_length": 4510.0, "completions/min_length": 3630.0, "completions/min_terminated_length": 3630.0, "epoch": 0.045442143122907105, "grad_norm": 0.0, "learning_rate": 8.45e-08, "loss": 0.0, "num_tokens": 10128978.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 905.0, "completions/max_terminated_length": 905.0, "completions/mean_length": 887.5, "completions/mean_terminated_length": 887.5, "completions/min_length": 870.0, "completions/min_terminated_length": 870.0, "epoch": 0.0454669477861838, "grad_norm": 0.0, "learning_rate": 8.4e-08, "loss": 0.0, "num_tokens": 10131581.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6051.0, "completions/mean_length": 7121.5, "completions/mean_terminated_length": 6051.0, "completions/min_length": 6051.0, "completions/min_terminated_length": 6051.0, "epoch": 0.0454917524494605, "grad_norm": 4.305724620819092, "learning_rate": 8.35e-08, "loss": -0.707, "num_tokens": 10138472.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1834 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.045516557112737194, "grad_norm": 0.0, "learning_rate": 8.3e-08, "loss": 0.0, "num_tokens": 10139396.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6482.0, "completions/max_terminated_length": 6482.0, "completions/mean_length": 6174.5, "completions/mean_terminated_length": 6174.5, "completions/min_length": 5867.0, "completions/min_terminated_length": 5867.0, "epoch": 0.04554136177601389, "grad_norm": 0.0, "learning_rate": 8.25e-08, "loss": 0.0, "num_tokens": 10152737.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4290.0, "completions/max_terminated_length": 4290.0, "completions/mean_length": 3657.0, "completions/mean_terminated_length": 3657.0, "completions/min_length": 3024.0, "completions/min_terminated_length": 3024.0, "epoch": 0.04556616643929059, "grad_norm": 0.0, "learning_rate": 8.2e-08, "loss": 0.0, "num_tokens": 10160945.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1837 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04559097110256728, "grad_norm": 0.0, "learning_rate": 8.15e-08, "loss": 0.0, "num_tokens": 10161811.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1838 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.045615775765843976, "grad_norm": 0.0, "learning_rate": 8.1e-08, "loss": 0.0, "num_tokens": 10162707.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1839 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04564058042912068, "grad_norm": 0.0, "learning_rate": 8.05e-08, "loss": 0.0, "num_tokens": 10163747.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1840 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04566538509239737, "grad_norm": 0.0, "learning_rate": 8e-08, "loss": 0.0, "num_tokens": 10164705.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6285.0, "completions/max_terminated_length": 6285.0, "completions/mean_length": 5989.5, "completions/mean_terminated_length": 5989.5, "completions/min_length": 5694.0, "completions/min_terminated_length": 5694.0, "epoch": 0.045690189755674064, "grad_norm": 3.2746148109436035, "learning_rate": 7.95e-08, "loss": 0.0349, "num_tokens": 10177622.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1745.0, "completions/max_terminated_length": 1745.0, "completions/mean_length": 1682.5, "completions/mean_terminated_length": 1682.5, "completions/min_length": 1620.0, "completions/min_terminated_length": 1620.0, "epoch": 0.045714994418950765, "grad_norm": 0.0, "learning_rate": 7.899999999999999e-08, "loss": 0.0, "num_tokens": 10181825.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5953.0, "completions/max_terminated_length": 5953.0, "completions/mean_length": 4506.5, "completions/mean_terminated_length": 4506.5, "completions/min_length": 3060.0, "completions/min_terminated_length": 3060.0, "epoch": 0.04573979908222746, "grad_norm": 0.0, "learning_rate": 7.85e-08, "loss": 0.0, "num_tokens": 10191754.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6685.0, "completions/mean_length": 7438.5, "completions/mean_terminated_length": 6685.0, "completions/min_length": 6685.0, "completions/min_terminated_length": 6685.0, "epoch": 0.04576460374550415, "grad_norm": 3.309657096862793, "learning_rate": 7.8e-08, "loss": -0.707, "num_tokens": 10199353.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4013.0, "completions/max_terminated_length": 4013.0, "completions/mean_length": 2716.0, "completions/mean_terminated_length": 2716.0, "completions/min_length": 1419.0, "completions/min_terminated_length": 1419.0, "epoch": 0.04578940840878085, "grad_norm": 0.0, "learning_rate": 7.75e-08, "loss": 0.0, "num_tokens": 10205635.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 902.0, "completions/max_terminated_length": 902.0, "completions/mean_length": 747.5, "completions/mean_terminated_length": 747.5, "completions/min_length": 593.0, "completions/min_terminated_length": 593.0, "epoch": 0.04581421307205755, "grad_norm": 0.0, "learning_rate": 7.7e-08, "loss": 0.0, "num_tokens": 10207960.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1847 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04583901773533424, "grad_norm": 0.0, "learning_rate": 7.649999999999999e-08, "loss": 0.0, "num_tokens": 10208928.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1848 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1252.0, "completions/max_terminated_length": 1252.0, "completions/mean_length": 1094.0, "completions/mean_terminated_length": 1094.0, "completions/min_length": 936.0, "completions/min_terminated_length": 936.0, "epoch": 0.04586382239861094, "grad_norm": 0.0, "learning_rate": 7.599999999999999e-08, "loss": 0.0, "num_tokens": 10212072.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4269.0, "completions/max_terminated_length": 4269.0, "completions/mean_length": 2726.0, "completions/mean_terminated_length": 2726.0, "completions/min_length": 1183.0, "completions/min_terminated_length": 1183.0, "epoch": 0.045888627061887635, "grad_norm": 3.9185147285461426, "learning_rate": 7.55e-08, "loss": 0.4002, "num_tokens": 10218388.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1850 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1380.0, "completions/mean_length": 4786.0, "completions/mean_terminated_length": 1380.0, "completions/min_length": 1380.0, "completions/min_terminated_length": 1380.0, "epoch": 0.04591343172516433, "grad_norm": 7.155021667480469, "learning_rate": 7.5e-08, "loss": -0.707, "num_tokens": 10220680.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1851 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4780.0, "completions/max_terminated_length": 4780.0, "completions/mean_length": 4196.0, "completions/mean_terminated_length": 4196.0, "completions/min_length": 3612.0, "completions/min_terminated_length": 3612.0, "epoch": 0.04593823638844103, "grad_norm": 0.0, "learning_rate": 7.45e-08, "loss": 0.0, "num_tokens": 10229884.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 487.5, "completions/mean_terminated_length": 487.5, "completions/min_length": 464.0, "completions/min_terminated_length": 464.0, "epoch": 0.045963041051717723, "grad_norm": 0.0, "learning_rate": 7.399999999999999e-08, "loss": 0.0, "num_tokens": 10231657.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1853 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04598784571499442, "grad_norm": 0.0, "learning_rate": 7.349999999999999e-08, "loss": 0.0, "num_tokens": 10232561.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5548.0, "completions/max_terminated_length": 5548.0, "completions/mean_length": 4972.5, "completions/mean_terminated_length": 4972.5, "completions/min_length": 4397.0, "completions/min_terminated_length": 4397.0, "epoch": 0.04601265037827112, "grad_norm": 0.0, "learning_rate": 7.299999999999999e-08, "loss": 0.0, "num_tokens": 10243366.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3419.0, "completions/max_terminated_length": 3419.0, "completions/mean_length": 2947.0, "completions/mean_terminated_length": 2947.0, "completions/min_length": 2475.0, "completions/min_terminated_length": 2475.0, "epoch": 0.04603745504154781, "grad_norm": 0.0, "learning_rate": 7.25e-08, "loss": 0.0, "num_tokens": 10250122.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5496.0, "completions/max_terminated_length": 5496.0, "completions/mean_length": 5180.5, "completions/mean_terminated_length": 5180.5, "completions/min_length": 4865.0, "completions/min_terminated_length": 4865.0, "epoch": 0.046062259704824506, "grad_norm": 0.0, "learning_rate": 7.2e-08, "loss": 0.0, "num_tokens": 10261333.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1857 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.046087064368101206, "grad_norm": 0.0, "learning_rate": 7.149999999999999e-08, "loss": 0.0, "num_tokens": 10262311.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4314.0, "completions/max_terminated_length": 4314.0, "completions/mean_length": 4098.5, "completions/mean_terminated_length": 4098.5, "completions/min_length": 3883.0, "completions/min_terminated_length": 3883.0, "epoch": 0.0461118690313779, "grad_norm": 0.0, "learning_rate": 7.099999999999999e-08, "loss": 0.0, "num_tokens": 10271374.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1744.0, "completions/max_terminated_length": 1744.0, "completions/mean_length": 1477.5, "completions/mean_terminated_length": 1477.5, "completions/min_length": 1211.0, "completions/min_terminated_length": 1211.0, "epoch": 0.046136673694654594, "grad_norm": 0.0, "learning_rate": 7.049999999999999e-08, "loss": 0.0, "num_tokens": 10275157.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6957.0, "completions/max_terminated_length": 6957.0, "completions/mean_length": 5646.5, "completions/mean_terminated_length": 5646.5, "completions/min_length": 4336.0, "completions/min_terminated_length": 4336.0, "epoch": 0.04616147835793129, "grad_norm": 0.0, "learning_rate": 7e-08, "loss": 0.0, "num_tokens": 10287312.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6577.0, "completions/max_terminated_length": 6577.0, "completions/mean_length": 3989.5, "completions/mean_terminated_length": 3989.5, "completions/min_length": 1402.0, "completions/min_terminated_length": 1402.0, "epoch": 0.04618628302120799, "grad_norm": 0.0, "learning_rate": 6.950000000000001e-08, "loss": 0.0, "num_tokens": 10296157.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4742.0, "completions/max_terminated_length": 4742.0, "completions/mean_length": 3764.0, "completions/mean_terminated_length": 3764.0, "completions/min_length": 2786.0, "completions/min_terminated_length": 2786.0, "epoch": 0.04621108768448468, "grad_norm": 0.0, "learning_rate": 6.900000000000001e-08, "loss": 0.0, "num_tokens": 10304555.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1133.0, "completions/max_terminated_length": 1133.0, "completions/mean_length": 1081.5, "completions/mean_terminated_length": 1081.5, "completions/min_length": 1030.0, "completions/min_terminated_length": 1030.0, "epoch": 0.046235892347761376, "grad_norm": 0.0, "learning_rate": 6.85e-08, "loss": 0.0, "num_tokens": 10307576.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4899.0, "completions/max_terminated_length": 4899.0, "completions/mean_length": 4506.0, "completions/mean_terminated_length": 4506.0, "completions/min_length": 4113.0, "completions/min_terminated_length": 4113.0, "epoch": 0.04626069701103808, "grad_norm": 2.884739637374878, "learning_rate": 6.8e-08, "loss": 0.0617, "num_tokens": 10317490.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1865 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04628550167431477, "grad_norm": 0.0, "learning_rate": 6.75e-08, "loss": 0.0, "num_tokens": 10318412.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 1177.0, "completions/mean_length": 4684.5, "completions/mean_terminated_length": 1177.0, "completions/min_length": 1177.0, "completions/min_terminated_length": 1177.0, "epoch": 0.046310306337591464, "grad_norm": 0.0, "learning_rate": 6.7e-08, "loss": 0.0, "num_tokens": 10320425.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1867 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.046335111000868165, "grad_norm": 0.0, "learning_rate": 6.65e-08, "loss": 0.0, "num_tokens": 10321407.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6123.0, "completions/max_terminated_length": 6123.0, "completions/mean_length": 5314.5, "completions/mean_terminated_length": 5314.5, "completions/min_length": 4506.0, "completions/min_terminated_length": 4506.0, "epoch": 0.04635991566414486, "grad_norm": 2.475740909576416, "learning_rate": 6.6e-08, "loss": 0.1076, "num_tokens": 10333200.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1869 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2072.0, "completions/max_terminated_length": 2072.0, "completions/mean_length": 1781.0, "completions/mean_terminated_length": 1781.0, "completions/min_length": 1490.0, "completions/min_terminated_length": 1490.0, "epoch": 0.04638472032742155, "grad_norm": 4.6034159660339355, "learning_rate": 6.55e-08, "loss": -0.1155, "num_tokens": 10337596.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1870 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6870.0, "completions/mean_length": 7531.0, "completions/mean_terminated_length": 6870.0, "completions/min_length": 6870.0, "completions/min_terminated_length": 6870.0, "epoch": 0.04640952499069825, "grad_norm": 0.0, "learning_rate": 6.5e-08, "loss": 0.0, "num_tokens": 10345278.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1871 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5871.0, "completions/mean_length": 7031.5, "completions/mean_terminated_length": 5871.0, "completions/min_length": 5871.0, "completions/min_terminated_length": 5871.0, "epoch": 0.04643432965397495, "grad_norm": 3.3990092277526855, "learning_rate": 6.45e-08, "loss": -0.707, "num_tokens": 10352063.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1872 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04645913431725164, "grad_norm": 0.0, "learning_rate": 6.4e-08, "loss": 0.0, "num_tokens": 10353025.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1873 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2490.0, "completions/max_terminated_length": 2490.0, "completions/mean_length": 2109.0, "completions/mean_terminated_length": 2109.0, "completions/min_length": 1728.0, "completions/min_terminated_length": 1728.0, "epoch": 0.04648393898052834, "grad_norm": 0.0, "learning_rate": 6.349999999999999e-08, "loss": 0.0, "num_tokens": 10358175.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4891.0, "completions/max_terminated_length": 4891.0, "completions/mean_length": 3792.5, "completions/mean_terminated_length": 3792.5, "completions/min_length": 2694.0, "completions/min_terminated_length": 2694.0, "epoch": 0.046508743643805035, "grad_norm": 0.0, "learning_rate": 6.3e-08, "loss": 0.0, "num_tokens": 10366796.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6629.0, "completions/max_terminated_length": 6629.0, "completions/mean_length": 5740.5, "completions/mean_terminated_length": 5740.5, "completions/min_length": 4852.0, "completions/min_terminated_length": 4852.0, "epoch": 0.04653354830708173, "grad_norm": 0.0, "learning_rate": 6.25e-08, "loss": 0.0, "num_tokens": 10379253.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1183.0, "completions/max_terminated_length": 1183.0, "completions/mean_length": 1163.0, "completions/mean_terminated_length": 1163.0, "completions/min_length": 1143.0, "completions/min_terminated_length": 1143.0, "epoch": 0.04655835297035843, "grad_norm": 0.0, "learning_rate": 6.2e-08, "loss": 0.0, "num_tokens": 10382425.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4437.0, "completions/mean_length": 6314.5, "completions/mean_terminated_length": 4437.0, "completions/min_length": 4437.0, "completions/min_terminated_length": 4437.0, "epoch": 0.046583157633635124, "grad_norm": 3.111818313598633, "learning_rate": 6.15e-08, "loss": -0.707, "num_tokens": 10387772.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1878 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3144.0, "completions/max_terminated_length": 3144.0, "completions/mean_length": 2427.0, "completions/mean_terminated_length": 2427.0, "completions/min_length": 1710.0, "completions/min_terminated_length": 1710.0, "epoch": 0.04660796229691182, "grad_norm": 0.0, "learning_rate": 6.099999999999999e-08, "loss": 0.0, "num_tokens": 10393492.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1879 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6891.0, "completions/max_terminated_length": 6891.0, "completions/mean_length": 6026.5, "completions/mean_terminated_length": 6026.5, "completions/min_length": 5162.0, "completions/min_terminated_length": 5162.0, "epoch": 0.04663276696018852, "grad_norm": 0.0, "learning_rate": 6.049999999999999e-08, "loss": 0.0, "num_tokens": 10406427.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1880 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04665757162346521, "grad_norm": 0.0, "learning_rate": 6e-08, "loss": 0.0, "num_tokens": 10407437.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1881 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4219.0, "completions/mean_length": 6205.5, "completions/mean_terminated_length": 4219.0, "completions/min_length": 4219.0, "completions/min_terminated_length": 4219.0, "epoch": 0.046682376286741906, "grad_norm": 0.0, "learning_rate": 5.95e-08, "loss": 0.0, "num_tokens": 10412658.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1882 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3856.0, "completions/max_terminated_length": 3856.0, "completions/mean_length": 3848.0, "completions/mean_terminated_length": 3848.0, "completions/min_length": 3840.0, "completions/min_terminated_length": 3840.0, "epoch": 0.046707180950018606, "grad_norm": 3.0552074909210205, "learning_rate": 5.899999999999999e-08, "loss": 0.0015, "num_tokens": 10421178.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1883 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0467319856132953, "grad_norm": 0.0, "learning_rate": 5.85e-08, "loss": 0.0, "num_tokens": 10422156.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4048.0, "completions/max_terminated_length": 4048.0, "completions/mean_length": 3357.0, "completions/mean_terminated_length": 3357.0, "completions/min_length": 2666.0, "completions/min_terminated_length": 2666.0, "epoch": 0.046756790276571994, "grad_norm": 2.7769646644592285, "learning_rate": 5.8e-08, "loss": 0.1455, "num_tokens": 10430008.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1885 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.046781594939848695, "grad_norm": 0.0, "learning_rate": 5.75e-08, "loss": 0.0, "num_tokens": 10431048.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2171.0, "completions/max_terminated_length": 2171.0, "completions/mean_length": 1451.0, "completions/mean_terminated_length": 1451.0, "completions/min_length": 731.0, "completions/min_terminated_length": 731.0, "epoch": 0.04680639960312539, "grad_norm": 0.0, "learning_rate": 5.7e-08, "loss": 0.0, "num_tokens": 10434778.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4318.0, "completions/max_terminated_length": 4318.0, "completions/mean_length": 3468.0, "completions/mean_terminated_length": 3468.0, "completions/min_length": 2618.0, "completions/min_terminated_length": 2618.0, "epoch": 0.04683120426640208, "grad_norm": 2.997753620147705, "learning_rate": 5.6499999999999996e-08, "loss": -0.1733, "num_tokens": 10442616.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1888 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04685600892967878, "grad_norm": 0.0, "learning_rate": 5.6e-08, "loss": 0.0, "num_tokens": 10443466.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5314.0, "completions/mean_length": 6753.0, "completions/mean_terminated_length": 5314.0, "completions/min_length": 5314.0, "completions/min_terminated_length": 5314.0, "epoch": 0.04688081359295548, "grad_norm": 4.1284050941467285, "learning_rate": 5.55e-08, "loss": -0.707, "num_tokens": 10449620.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6025.0, "completions/mean_length": 7108.5, "completions/mean_terminated_length": 6025.0, "completions/min_length": 6025.0, "completions/min_terminated_length": 6025.0, "epoch": 0.04690561825623217, "grad_norm": 0.0, "learning_rate": 5.4999999999999996e-08, "loss": 0.0, "num_tokens": 10456513.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1891 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.046930422919508864, "grad_norm": 0.0, "learning_rate": 5.45e-08, "loss": 0.0, "num_tokens": 10457385.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1892 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.046955227582785565, "grad_norm": 0.0, "learning_rate": 5.3999999999999994e-08, "loss": 0.0, "num_tokens": 10458271.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3358.0, "completions/max_terminated_length": 3358.0, "completions/mean_length": 2819.0, "completions/mean_terminated_length": 2819.0, "completions/min_length": 2280.0, "completions/min_terminated_length": 2280.0, "epoch": 0.04698003224606226, "grad_norm": 0.0, "learning_rate": 5.3499999999999996e-08, "loss": 0.0, "num_tokens": 10464713.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1894 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04700483690933895, "grad_norm": 0.0, "learning_rate": 5.3e-08, "loss": 0.0, "num_tokens": 10465583.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1171.0, "completions/max_terminated_length": 1171.0, "completions/mean_length": 1076.5, "completions/mean_terminated_length": 1076.5, "completions/min_length": 982.0, "completions/min_terminated_length": 982.0, "epoch": 0.04702964157261565, "grad_norm": 0.0, "learning_rate": 5.2499999999999994e-08, "loss": 0.0, "num_tokens": 10468624.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 682.0, "completions/max_terminated_length": 682.0, "completions/mean_length": 627.5, "completions/mean_terminated_length": 627.5, "completions/min_length": 573.0, "completions/min_terminated_length": 573.0, "epoch": 0.04705444623589235, "grad_norm": 0.0, "learning_rate": 5.1999999999999996e-08, "loss": 0.0, "num_tokens": 10470731.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1897 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 907.0, "completions/max_terminated_length": 907.0, "completions/mean_length": 798.5, "completions/mean_terminated_length": 798.5, "completions/min_length": 690.0, "completions/min_terminated_length": 690.0, "epoch": 0.04707925089916904, "grad_norm": 0.0, "learning_rate": 5.15e-08, "loss": 0.0, "num_tokens": 10473200.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1898 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 762.0, "completions/max_terminated_length": 762.0, "completions/mean_length": 578.5, "completions/mean_terminated_length": 578.5, "completions/min_length": 395.0, "completions/min_terminated_length": 395.0, "epoch": 0.04710405556244574, "grad_norm": 0.0, "learning_rate": 5.0999999999999993e-08, "loss": 0.0, "num_tokens": 10475221.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1899 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.047128860225722435, "grad_norm": 0.0, "learning_rate": 5.05e-08, "loss": 0.0, "num_tokens": 10476097.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 2556.0, "completions/mean_length": 5374.0, "completions/mean_terminated_length": 2556.0, "completions/min_length": 2556.0, "completions/min_terminated_length": 2556.0, "epoch": 0.04715366488899913, "grad_norm": 3.972573757171631, "learning_rate": 5e-08, "loss": -0.707, "num_tokens": 10479665.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1931.0, "completions/max_terminated_length": 1931.0, "completions/mean_length": 1783.0, "completions/mean_terminated_length": 1783.0, "completions/min_length": 1635.0, "completions/min_terminated_length": 1635.0, "epoch": 0.04717846955227583, "grad_norm": 0.0, "learning_rate": 4.95e-08, "loss": 0.0, "num_tokens": 10484063.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6195.0, "completions/max_terminated_length": 6195.0, "completions/mean_length": 3932.0, "completions/mean_terminated_length": 3932.0, "completions/min_length": 1669.0, "completions/min_terminated_length": 1669.0, "epoch": 0.047203274215552524, "grad_norm": 0.0, "learning_rate": 4.9e-08, "loss": 0.0, "num_tokens": 10492793.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1903 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04722807887882922, "grad_norm": 0.0, "learning_rate": 4.85e-08, "loss": 0.0, "num_tokens": 10493747.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7494.0, "completions/mean_length": 7843.0, "completions/mean_terminated_length": 7494.0, "completions/min_length": 7494.0, "completions/min_terminated_length": 7494.0, "epoch": 0.04725288354210592, "grad_norm": 3.4809465408325195, "learning_rate": 4.8e-08, "loss": -0.707, "num_tokens": 10502205.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7931.0, "completions/max_terminated_length": 7931.0, "completions/mean_length": 5781.5, "completions/mean_terminated_length": 5781.5, "completions/min_length": 3632.0, "completions/min_terminated_length": 3632.0, "epoch": 0.04727768820538261, "grad_norm": 2.007009744644165, "learning_rate": 4.7499999999999995e-08, "loss": 0.2629, "num_tokens": 10514606.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2875.0, "completions/max_terminated_length": 2875.0, "completions/mean_length": 2392.0, "completions/mean_terminated_length": 2392.0, "completions/min_length": 1909.0, "completions/min_terminated_length": 1909.0, "epoch": 0.047302492868659306, "grad_norm": 0.0, "learning_rate": 4.7e-08, "loss": 0.0, "num_tokens": 10520264.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3013.0, "completions/max_terminated_length": 3013.0, "completions/mean_length": 2072.5, "completions/mean_terminated_length": 2072.5, "completions/min_length": 1132.0, "completions/min_terminated_length": 1132.0, "epoch": 0.047327297531936006, "grad_norm": 0.0, "learning_rate": 4.65e-08, "loss": 0.0, "num_tokens": 10525273.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6029.0, "completions/max_terminated_length": 6029.0, "completions/mean_length": 5332.0, "completions/mean_terminated_length": 5332.0, "completions/min_length": 4635.0, "completions/min_terminated_length": 4635.0, "epoch": 0.0473521021952127, "grad_norm": 2.4341959953308105, "learning_rate": 4.5999999999999995e-08, "loss": -0.0924, "num_tokens": 10536839.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1909 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4667.0, "completions/max_terminated_length": 4667.0, "completions/mean_length": 3396.5, "completions/mean_terminated_length": 3396.5, "completions/min_length": 2126.0, "completions/min_terminated_length": 2126.0, "epoch": 0.047376906858489394, "grad_norm": 3.1551597118377686, "learning_rate": 4.55e-08, "loss": 0.2645, "num_tokens": 10544644.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6898.0, "completions/mean_length": 7545.0, "completions/mean_terminated_length": 6898.0, "completions/min_length": 6898.0, "completions/min_terminated_length": 6898.0, "epoch": 0.047401711521766095, "grad_norm": 0.0, "learning_rate": 4.5e-08, "loss": 0.0, "num_tokens": 10552518.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1911 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2939.0, "completions/max_terminated_length": 2939.0, "completions/mean_length": 2309.5, "completions/mean_terminated_length": 2309.5, "completions/min_length": 1680.0, "completions/min_terminated_length": 1680.0, "epoch": 0.04742651618504279, "grad_norm": 0.0, "learning_rate": 4.4499999999999995e-08, "loss": 0.0, "num_tokens": 10558075.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3179.0, "completions/max_terminated_length": 3179.0, "completions/mean_length": 2522.0, "completions/mean_terminated_length": 2522.0, "completions/min_length": 1865.0, "completions/min_terminated_length": 1865.0, "epoch": 0.04745132084831948, "grad_norm": 0.0, "learning_rate": 4.4e-08, "loss": 0.0, "num_tokens": 10564063.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4457.0, "completions/max_terminated_length": 4457.0, "completions/mean_length": 4357.0, "completions/mean_terminated_length": 4357.0, "completions/min_length": 4257.0, "completions/min_terminated_length": 4257.0, "epoch": 0.04747612551159618, "grad_norm": 0.0, "learning_rate": 4.349999999999999e-08, "loss": 0.0, "num_tokens": 10573647.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1223.0, "completions/max_terminated_length": 1223.0, "completions/mean_length": 1037.0, "completions/mean_terminated_length": 1037.0, "completions/min_length": 851.0, "completions/min_terminated_length": 851.0, "epoch": 0.04750093017487288, "grad_norm": 0.0, "learning_rate": 4.2999999999999995e-08, "loss": 0.0, "num_tokens": 10576631.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 515.0, "completions/max_terminated_length": 515.0, "completions/mean_length": 486.5, "completions/mean_terminated_length": 486.5, "completions/min_length": 458.0, "completions/min_terminated_length": 458.0, "epoch": 0.04752573483814957, "grad_norm": 0.0, "learning_rate": 4.2500000000000003e-08, "loss": 0.0, "num_tokens": 10578398.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1338.0, "completions/max_terminated_length": 1338.0, "completions/mean_length": 994.5, "completions/mean_terminated_length": 994.5, "completions/min_length": 651.0, "completions/min_terminated_length": 651.0, "epoch": 0.04755053950142627, "grad_norm": 0.0, "learning_rate": 4.2e-08, "loss": 0.0, "num_tokens": 10581229.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1917 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 586.0, "completions/max_terminated_length": 586.0, "completions/mean_length": 546.0, "completions/mean_terminated_length": 546.0, "completions/min_length": 506.0, "completions/min_terminated_length": 506.0, "epoch": 0.047575344164702965, "grad_norm": 0.0, "learning_rate": 4.15e-08, "loss": 0.0, "num_tokens": 10583197.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7425.0, "completions/mean_length": 7808.5, "completions/mean_terminated_length": 7425.0, "completions/min_length": 7425.0, "completions/min_terminated_length": 7425.0, "epoch": 0.04760014882797966, "grad_norm": 0.0, "learning_rate": 4.1e-08, "loss": 0.0, "num_tokens": 10591640.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1919 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04762495349125636, "grad_norm": 0.0, "learning_rate": 4.05e-08, "loss": 0.0, "num_tokens": 10592696.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1920 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3329.0, "completions/max_terminated_length": 3329.0, "completions/mean_length": 2024.0, "completions/mean_terminated_length": 2024.0, "completions/min_length": 719.0, "completions/min_terminated_length": 719.0, "epoch": 0.04764975815453305, "grad_norm": 0.0, "learning_rate": 4e-08, "loss": 0.0, "num_tokens": 10597594.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7569.0, "completions/max_terminated_length": 7569.0, "completions/mean_length": 6962.0, "completions/mean_terminated_length": 6962.0, "completions/min_length": 6355.0, "completions/min_terminated_length": 6355.0, "epoch": 0.04767456281780975, "grad_norm": 0.0, "learning_rate": 3.9499999999999996e-08, "loss": 0.0, "num_tokens": 10612444.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1922 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6470.0, "completions/mean_length": 7331.0, "completions/mean_terminated_length": 6470.0, "completions/min_length": 6470.0, "completions/min_terminated_length": 6470.0, "epoch": 0.04769936748108644, "grad_norm": 3.5158562660217285, "learning_rate": 3.9e-08, "loss": -0.707, "num_tokens": 10619770.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1923 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04772417214436314, "grad_norm": 0.0, "learning_rate": 3.85e-08, "loss": 0.0, "num_tokens": 10620610.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4942.0, "completions/max_terminated_length": 4942.0, "completions/mean_length": 3342.0, "completions/mean_terminated_length": 3342.0, "completions/min_length": 1742.0, "completions/min_terminated_length": 1742.0, "epoch": 0.047748976807639835, "grad_norm": 0.0, "learning_rate": 3.7999999999999996e-08, "loss": 0.0, "num_tokens": 10628168.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2056.0, "completions/max_terminated_length": 2056.0, "completions/mean_length": 1859.5, "completions/mean_terminated_length": 1859.5, "completions/min_length": 1663.0, "completions/min_terminated_length": 1663.0, "epoch": 0.04777378147091653, "grad_norm": 0.0, "learning_rate": 3.75e-08, "loss": 0.0, "num_tokens": 10632689.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1926 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04779858613419323, "grad_norm": 0.0, "learning_rate": 3.6999999999999994e-08, "loss": 0.0, "num_tokens": 10633561.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1927 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 808.0, "completions/max_terminated_length": 808.0, "completions/mean_length": 761.5, "completions/mean_terminated_length": 761.5, "completions/min_length": 715.0, "completions/min_terminated_length": 715.0, "epoch": 0.047823390797469924, "grad_norm": 0.0, "learning_rate": 3.6499999999999996e-08, "loss": 0.0, "num_tokens": 10635928.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1890.0, "completions/max_terminated_length": 1890.0, "completions/mean_length": 1770.5, "completions/mean_terminated_length": 1770.5, "completions/min_length": 1651.0, "completions/min_terminated_length": 1651.0, "epoch": 0.04784819546074662, "grad_norm": 0.0, "learning_rate": 3.6e-08, "loss": 0.0, "num_tokens": 10640385.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 817.0, "completions/max_terminated_length": 817.0, "completions/mean_length": 736.0, "completions/mean_terminated_length": 736.0, "completions/min_length": 655.0, "completions/min_terminated_length": 655.0, "epoch": 0.04787300012402332, "grad_norm": 0.0, "learning_rate": 3.5499999999999994e-08, "loss": 0.0, "num_tokens": 10642673.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 6174.0, "completions/mean_length": 7183.0, "completions/mean_terminated_length": 6174.0, "completions/min_length": 6174.0, "completions/min_terminated_length": 6174.0, "epoch": 0.04789780478730001, "grad_norm": 3.3348278999328613, "learning_rate": 3.5e-08, "loss": -0.707, "num_tokens": 10649641.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4769.0, "completions/max_terminated_length": 4769.0, "completions/mean_length": 3545.0, "completions/mean_terminated_length": 3545.0, "completions/min_length": 2321.0, "completions/min_terminated_length": 2321.0, "epoch": 0.047922609450576706, "grad_norm": 3.1587796211242676, "learning_rate": 3.4500000000000005e-08, "loss": -0.2441, "num_tokens": 10657595.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1932 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04794741411385341, "grad_norm": 0.0, "learning_rate": 3.4e-08, "loss": 0.0, "num_tokens": 10658489.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2479.0, "completions/max_terminated_length": 2479.0, "completions/mean_length": 2087.0, "completions/mean_terminated_length": 2087.0, "completions/min_length": 1695.0, "completions/min_terminated_length": 1695.0, "epoch": 0.0479722187771301, "grad_norm": 0.0, "learning_rate": 3.35e-08, "loss": 0.0, "num_tokens": 10663481.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1161.0, "completions/max_terminated_length": 1161.0, "completions/mean_length": 901.0, "completions/mean_terminated_length": 901.0, "completions/min_length": 641.0, "completions/min_terminated_length": 641.0, "epoch": 0.047997023440406794, "grad_norm": 0.0, "learning_rate": 3.3e-08, "loss": 0.0, "num_tokens": 10666111.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4034.0, "completions/max_terminated_length": 4034.0, "completions/mean_length": 3279.0, "completions/mean_terminated_length": 3279.0, "completions/min_length": 2524.0, "completions/min_terminated_length": 2524.0, "epoch": 0.048021828103683495, "grad_norm": 4.116697311401367, "learning_rate": 3.25e-08, "loss": 0.1628, "num_tokens": 10673519.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1031.0, "completions/max_terminated_length": 1031.0, "completions/mean_length": 843.5, "completions/mean_terminated_length": 843.5, "completions/min_length": 656.0, "completions/min_terminated_length": 656.0, "epoch": 0.04804663276696019, "grad_norm": 0.0, "learning_rate": 3.2e-08, "loss": 0.0, "num_tokens": 10676000.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7261.0, "completions/max_terminated_length": 7261.0, "completions/mean_length": 5342.5, "completions/mean_terminated_length": 5342.5, "completions/min_length": 3424.0, "completions/min_terminated_length": 3424.0, "epoch": 0.04807143743023688, "grad_norm": 0.0, "learning_rate": 3.15e-08, "loss": 0.0, "num_tokens": 10687539.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2832.0, "completions/max_terminated_length": 2832.0, "completions/mean_length": 2818.5, "completions/mean_terminated_length": 2818.5, "completions/min_length": 2805.0, "completions/min_terminated_length": 2805.0, "epoch": 0.04809624209351358, "grad_norm": 0.0, "learning_rate": 3.1e-08, "loss": 0.0, "num_tokens": 10694072.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1939 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04812104675679028, "grad_norm": 0.0, "learning_rate": 3.0499999999999995e-08, "loss": 0.0, "num_tokens": 10695068.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1940 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04814585142006697, "grad_norm": 0.0, "learning_rate": 3e-08, "loss": 0.0, "num_tokens": 10696092.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1941 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04817065608334367, "grad_norm": 0.0, "learning_rate": 2.9499999999999996e-08, "loss": 0.0, "num_tokens": 10697434.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1942 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4364.0, "completions/mean_length": 6278.0, "completions/mean_terminated_length": 4364.0, "completions/min_length": 4364.0, "completions/min_terminated_length": 4364.0, "epoch": 0.048195460746620365, "grad_norm": 0.0, "learning_rate": 2.9e-08, "loss": 0.0, "num_tokens": 10702712.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1943 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04822026540989706, "grad_norm": 0.0, "learning_rate": 2.85e-08, "loss": 0.0, "num_tokens": 10703808.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7335.0, "completions/max_terminated_length": 7335.0, "completions/mean_length": 4446.5, "completions/mean_terminated_length": 4446.5, "completions/min_length": 1558.0, "completions/min_terminated_length": 1558.0, "epoch": 0.04824507007317376, "grad_norm": 0.0, "learning_rate": 2.8e-08, "loss": 0.0, "num_tokens": 10713547.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 4093.0, "completions/mean_length": 6142.5, "completions/mean_terminated_length": 4093.0, "completions/min_length": 4093.0, "completions/min_terminated_length": 4093.0, "epoch": 0.048269874736450454, "grad_norm": 4.024594783782959, "learning_rate": 2.7499999999999998e-08, "loss": -0.707, "num_tokens": 10718566.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5870.0, "completions/max_terminated_length": 5870.0, "completions/mean_length": 5778.5, "completions/mean_terminated_length": 5778.5, "completions/min_length": 5687.0, "completions/min_terminated_length": 5687.0, "epoch": 0.04829467939972715, "grad_norm": 0.0, "learning_rate": 2.6999999999999997e-08, "loss": 0.0, "num_tokens": 10730949.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1947 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5975.0, "completions/max_terminated_length": 5975.0, "completions/mean_length": 5347.0, "completions/mean_terminated_length": 5347.0, "completions/min_length": 4719.0, "completions/min_terminated_length": 4719.0, "epoch": 0.04831948406300385, "grad_norm": 0.0, "learning_rate": 2.65e-08, "loss": 0.0, "num_tokens": 10742563.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1948 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1902.0, "completions/max_terminated_length": 1902.0, "completions/mean_length": 1791.0, "completions/mean_terminated_length": 1791.0, "completions/min_length": 1680.0, "completions/min_terminated_length": 1680.0, "epoch": 0.04834428872628054, "grad_norm": 0.0, "learning_rate": 2.5999999999999998e-08, "loss": 0.0, "num_tokens": 10746981.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1949 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.048369093389557236, "grad_norm": 0.0, "learning_rate": 2.5499999999999997e-08, "loss": 0.0, "num_tokens": 10747813.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6387.0, "completions/max_terminated_length": 6387.0, "completions/mean_length": 4674.0, "completions/mean_terminated_length": 4674.0, "completions/min_length": 2961.0, "completions/min_terminated_length": 2961.0, "epoch": 0.04839389805283393, "grad_norm": 0.0, "learning_rate": 2.5e-08, "loss": 0.0, "num_tokens": 10758125.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5052.0, "completions/mean_length": 6622.0, "completions/mean_terminated_length": 5052.0, "completions/min_length": 5052.0, "completions/min_terminated_length": 5052.0, "epoch": 0.04841870271611063, "grad_norm": 3.9684360027313232, "learning_rate": 2.45e-08, "loss": -0.707, "num_tokens": 10764053.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1481.0, "completions/max_terminated_length": 1481.0, "completions/mean_length": 1161.5, "completions/mean_terminated_length": 1161.5, "completions/min_length": 842.0, "completions/min_terminated_length": 842.0, "epoch": 0.048443507379387324, "grad_norm": 0.0, "learning_rate": 2.4e-08, "loss": 0.0, "num_tokens": 10767282.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5988.0, "completions/mean_length": 7090.0, "completions/mean_terminated_length": 5988.0, "completions/min_length": 5988.0, "completions/min_terminated_length": 5988.0, "epoch": 0.04846831204266402, "grad_norm": 3.166559934616089, "learning_rate": 2.35e-08, "loss": -0.707, "num_tokens": 10774214.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2160.0, "completions/max_terminated_length": 2160.0, "completions/mean_length": 1871.0, "completions/mean_terminated_length": 1871.0, "completions/min_length": 1582.0, "completions/min_terminated_length": 1582.0, "epoch": 0.04849311670594072, "grad_norm": 0.0, "learning_rate": 2.2999999999999998e-08, "loss": 0.0, "num_tokens": 10778900.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2070.0, "completions/max_terminated_length": 2070.0, "completions/mean_length": 1903.5, "completions/mean_terminated_length": 1903.5, "completions/min_length": 1737.0, "completions/min_terminated_length": 1737.0, "epoch": 0.04851792136921741, "grad_norm": 0.0, "learning_rate": 2.25e-08, "loss": 0.0, "num_tokens": 10783555.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2097.0, "completions/max_terminated_length": 2097.0, "completions/mean_length": 1748.5, "completions/mean_terminated_length": 1748.5, "completions/min_length": 1400.0, "completions/min_terminated_length": 1400.0, "epoch": 0.048542726032494106, "grad_norm": 0.0, "learning_rate": 2.2e-08, "loss": 0.0, "num_tokens": 10787856.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5633.0, "completions/mean_length": 6912.5, "completions/mean_terminated_length": 5633.0, "completions/min_length": 5633.0, "completions/min_terminated_length": 5633.0, "epoch": 0.04856753069577081, "grad_norm": 0.0, "learning_rate": 2.1499999999999997e-08, "loss": 0.0, "num_tokens": 10794513.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7623.0, "completions/max_terminated_length": 7623.0, "completions/mean_length": 6811.0, "completions/mean_terminated_length": 6811.0, "completions/min_length": 5999.0, "completions/min_terminated_length": 5999.0, "epoch": 0.0485923353590475, "grad_norm": 0.0, "learning_rate": 2.1e-08, "loss": 0.0, "num_tokens": 10809017.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4294.0, "completions/max_terminated_length": 4294.0, "completions/mean_length": 4170.0, "completions/mean_terminated_length": 4170.0, "completions/min_length": 4046.0, "completions/min_terminated_length": 4046.0, "epoch": 0.048617140022324194, "grad_norm": 0.0, "learning_rate": 2.05e-08, "loss": 0.0, "num_tokens": 10818185.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1960 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.048641944685600895, "grad_norm": 0.0, "learning_rate": 2e-08, "loss": 0.0, "num_tokens": 10819151.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1493.0, "completions/max_terminated_length": 1493.0, "completions/mean_length": 1389.0, "completions/mean_terminated_length": 1389.0, "completions/min_length": 1285.0, "completions/min_terminated_length": 1285.0, "epoch": 0.04866674934887759, "grad_norm": 4.417654991149902, "learning_rate": 1.95e-08, "loss": 0.0529, "num_tokens": 10822763.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 8090.0, "completions/max_terminated_length": 8090.0, "completions/mean_length": 6495.5, "completions/mean_terminated_length": 6495.5, "completions/min_length": 4901.0, "completions/min_terminated_length": 4901.0, "epoch": 0.04869155401215428, "grad_norm": 2.1013283729553223, "learning_rate": 1.8999999999999998e-08, "loss": 0.1736, "num_tokens": 10836634.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2209.0, "completions/max_terminated_length": 2209.0, "completions/mean_length": 1512.0, "completions/mean_terminated_length": 1512.0, "completions/min_length": 815.0, "completions/min_terminated_length": 815.0, "epoch": 0.04871635867543098, "grad_norm": 0.0, "learning_rate": 1.8499999999999997e-08, "loss": 0.0, "num_tokens": 10840544.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6582.0, "completions/max_terminated_length": 6582.0, "completions/mean_length": 6233.5, "completions/mean_terminated_length": 6233.5, "completions/min_length": 5885.0, "completions/min_terminated_length": 5885.0, "epoch": 0.04874116333870768, "grad_norm": 0.0, "learning_rate": 1.8e-08, "loss": 0.0, "num_tokens": 10854047.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1965 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04876596800198437, "grad_norm": 0.0, "learning_rate": 1.75e-08, "loss": 0.0, "num_tokens": 10855113.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2298.0, "completions/max_terminated_length": 2298.0, "completions/mean_length": 2014.5, "completions/mean_terminated_length": 2014.5, "completions/min_length": 1731.0, "completions/min_terminated_length": 1731.0, "epoch": 0.04879077266526107, "grad_norm": 0.0, "learning_rate": 1.7e-08, "loss": 0.0, "num_tokens": 10859978.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1967 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 5226.0, "completions/max_terminated_length": 5226.0, "completions/mean_length": 4335.0, "completions/mean_terminated_length": 4335.0, "completions/min_length": 3444.0, "completions/min_terminated_length": 3444.0, "epoch": 0.048815577328537765, "grad_norm": 0.0, "learning_rate": 1.65e-08, "loss": 0.0, "num_tokens": 10869488.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 520.0, "completions/max_terminated_length": 520.0, "completions/mean_length": 476.0, "completions/mean_terminated_length": 476.0, "completions/min_length": 432.0, "completions/min_terminated_length": 432.0, "epoch": 0.04884038199181446, "grad_norm": 0.0, "learning_rate": 1.6e-08, "loss": 0.0, "num_tokens": 10871264.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6494.0, "completions/max_terminated_length": 6494.0, "completions/mean_length": 4637.0, "completions/mean_terminated_length": 4637.0, "completions/min_length": 2780.0, "completions/min_terminated_length": 2780.0, "epoch": 0.04886518665509116, "grad_norm": 0.0, "learning_rate": 1.55e-08, "loss": 0.0, "num_tokens": 10881426.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1970 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5006.0, "completions/mean_length": 6599.0, "completions/mean_terminated_length": 5006.0, "completions/min_length": 5006.0, "completions/min_terminated_length": 5006.0, "epoch": 0.048889991318367854, "grad_norm": 4.146580219268799, "learning_rate": 1.5e-08, "loss": -0.707, "num_tokens": 10887328.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1971 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04891479598164455, "grad_norm": 0.0, "learning_rate": 1.45e-08, "loss": 0.0, "num_tokens": 10888174.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 804.0, "completions/max_terminated_length": 804.0, "completions/mean_length": 678.5, "completions/mean_terminated_length": 678.5, "completions/min_length": 553.0, "completions/min_terminated_length": 553.0, "epoch": 0.04893960064492125, "grad_norm": 0.0, "learning_rate": 1.4e-08, "loss": 0.0, "num_tokens": 10890357.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 7321.0, "completions/mean_length": 7756.5, "completions/mean_terminated_length": 7321.0, "completions/min_length": 7321.0, "completions/min_terminated_length": 7321.0, "epoch": 0.04896440530819794, "grad_norm": 0.0, "learning_rate": 1.3499999999999998e-08, "loss": 0.0, "num_tokens": 10898508.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1974 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.048989209971474636, "grad_norm": 0.0, "learning_rate": 1.2999999999999999e-08, "loss": 0.0, "num_tokens": 10899526.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1234.0, "completions/max_terminated_length": 1234.0, "completions/mean_length": 1010.0, "completions/mean_terminated_length": 1010.0, "completions/min_length": 786.0, "completions/min_terminated_length": 786.0, "epoch": 0.049014014634751336, "grad_norm": 0.0, "learning_rate": 1.25e-08, "loss": 0.0, "num_tokens": 10902340.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2574.0, "completions/max_terminated_length": 2574.0, "completions/mean_length": 2363.0, "completions/mean_terminated_length": 2363.0, "completions/min_length": 2152.0, "completions/min_terminated_length": 2152.0, "epoch": 0.04903881929802803, "grad_norm": 0.0, "learning_rate": 1.2e-08, "loss": 0.0, "num_tokens": 10907964.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1977 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.049063623961304724, "grad_norm": 0.0, "learning_rate": 1.1499999999999999e-08, "loss": 0.0, "num_tokens": 10908960.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1406.0, "completions/max_terminated_length": 1406.0, "completions/mean_length": 1138.5, "completions/mean_terminated_length": 1138.5, "completions/min_length": 871.0, "completions/min_terminated_length": 871.0, "epoch": 0.049088428624581425, "grad_norm": 0.0, "learning_rate": 1.1e-08, "loss": 0.0, "num_tokens": 10912069.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1979 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4210.0, "completions/max_terminated_length": 4210.0, "completions/mean_length": 3082.0, "completions/mean_terminated_length": 3082.0, "completions/min_length": 1954.0, "completions/min_terminated_length": 1954.0, "epoch": 0.04911323328785812, "grad_norm": 0.0, "learning_rate": 1.05e-08, "loss": 0.0, "num_tokens": 10919095.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 685.0, "completions/max_terminated_length": 685.0, "completions/mean_length": 529.5, "completions/mean_terminated_length": 529.5, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "epoch": 0.04913803795113481, "grad_norm": 0.0, "learning_rate": 1e-08, "loss": 0.0, "num_tokens": 10920962.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1877.0, "completions/max_terminated_length": 1877.0, "completions/mean_length": 1658.5, "completions/mean_terminated_length": 1658.5, "completions/min_length": 1440.0, "completions/min_terminated_length": 1440.0, "epoch": 0.049162842614411506, "grad_norm": 0.0, "learning_rate": 9.499999999999999e-09, "loss": 0.0, "num_tokens": 10925167.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1982 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.04918764727768821, "grad_norm": 0.0, "learning_rate": 9e-09, "loss": 0.0, "num_tokens": 10926063.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1983 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.0492124519409649, "grad_norm": 0.0, "learning_rate": 8.5e-09, "loss": 0.0, "num_tokens": 10926997.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1984 }, { "clip_ratio/high_max": NaN, "clip_ratio/high_mean": NaN, "clip_ratio/low_mean": NaN, "clip_ratio/low_min": NaN, "clip_ratio/region_mean": NaN, "completions/clipped_ratio": 1.0, "completions/max_length": 8192.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 8192.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 8192.0, "completions/min_terminated_length": 0.0, "epoch": 0.049237256604241594, "grad_norm": 0.0, "learning_rate": 8e-09, "loss": 0.0, "num_tokens": 10927969.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 890.0, "completions/max_terminated_length": 890.0, "completions/mean_length": 808.0, "completions/mean_terminated_length": 808.0, "completions/min_length": 726.0, "completions/min_terminated_length": 726.0, "epoch": 0.049262061267518295, "grad_norm": 0.0, "learning_rate": 7.5e-09, "loss": 0.0, "num_tokens": 10930427.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1404.0, "completions/max_terminated_length": 1404.0, "completions/mean_length": 1235.5, "completions/mean_terminated_length": 1235.5, "completions/min_length": 1067.0, "completions/min_terminated_length": 1067.0, "epoch": 0.04928686593079499, "grad_norm": 0.0, "learning_rate": 7e-09, "loss": 0.0, "num_tokens": 10933704.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1942.0, "completions/max_terminated_length": 1942.0, "completions/mean_length": 1734.0, "completions/mean_terminated_length": 1734.0, "completions/min_length": 1526.0, "completions/min_terminated_length": 1526.0, "epoch": 0.04931167059407168, "grad_norm": 0.0, "learning_rate": 6.4999999999999995e-09, "loss": 0.0, "num_tokens": 10938042.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1365.0, "completions/max_terminated_length": 1365.0, "completions/mean_length": 1231.5, "completions/mean_terminated_length": 1231.5, "completions/min_length": 1098.0, "completions/min_terminated_length": 1098.0, "epoch": 0.04933647525734838, "grad_norm": 0.0, "learning_rate": 6e-09, "loss": 0.0, "num_tokens": 10941345.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 788.0, "completions/max_terminated_length": 788.0, "completions/mean_length": 755.5, "completions/mean_terminated_length": 755.5, "completions/min_length": 723.0, "completions/min_terminated_length": 723.0, "epoch": 0.04936127992062508, "grad_norm": 0.0, "learning_rate": 5.5e-09, "loss": 0.0, "num_tokens": 10943656.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1990 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5812.0, "completions/mean_length": 7002.0, "completions/mean_terminated_length": 5812.0, "completions/min_length": 5812.0, "completions/min_terminated_length": 5812.0, "epoch": 0.04938608458390177, "grad_norm": 2.7499821186065674, "learning_rate": 5e-09, "loss": -0.707, "num_tokens": 10950304.0, "reward": 0.5, "reward_std": 0.7071067690849304, "rewards/accuracy_reward/mean": 0.5, "rewards/accuracy_reward/std": 0.7071067690849304, "step": 1991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4117.0, "completions/max_terminated_length": 4117.0, "completions/mean_length": 3134.0, "completions/mean_terminated_length": 3134.0, "completions/min_length": 2151.0, "completions/min_terminated_length": 2151.0, "epoch": 0.04941088924717847, "grad_norm": 0.0, "learning_rate": 4.5e-09, "loss": 0.0, "num_tokens": 10957486.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 3597.0, "completions/max_terminated_length": 3597.0, "completions/mean_length": 3145.0, "completions/mean_terminated_length": 3145.0, "completions/min_length": 2693.0, "completions/min_terminated_length": 2693.0, "epoch": 0.049435693910455165, "grad_norm": 0.0, "learning_rate": 4e-09, "loss": 0.0, "num_tokens": 10964592.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 624.0, "completions/max_terminated_length": 624.0, "completions/mean_length": 612.0, "completions/mean_terminated_length": 612.0, "completions/min_length": 600.0, "completions/min_terminated_length": 600.0, "epoch": 0.04946049857373186, "grad_norm": 0.0, "learning_rate": 3.5e-09, "loss": 0.0, "num_tokens": 10966640.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1314.0, "completions/max_terminated_length": 1314.0, "completions/mean_length": 1143.0, "completions/mean_terminated_length": 1143.0, "completions/min_length": 972.0, "completions/min_terminated_length": 972.0, "epoch": 0.04948530323700856, "grad_norm": 0.0, "learning_rate": 3e-09, "loss": 0.0, "num_tokens": 10969858.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 7330.0, "completions/max_terminated_length": 7330.0, "completions/mean_length": 5171.5, "completions/mean_terminated_length": 5171.5, "completions/min_length": 3013.0, "completions/min_terminated_length": 3013.0, "epoch": 0.049510107900285254, "grad_norm": 0.0, "learning_rate": 2.5e-09, "loss": 0.0, "num_tokens": 10981037.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 8192.0, "completions/max_terminated_length": 5904.0, "completions/mean_length": 7048.0, "completions/mean_terminated_length": 5904.0, "completions/min_length": 5904.0, "completions/min_terminated_length": 5904.0, "epoch": 0.04953491256356195, "grad_norm": 0.0, "learning_rate": 2e-09, "loss": 0.0, "num_tokens": 10987833.0, "reward": 0.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 0.0, "rewards/accuracy_reward/std": 0.0, "step": 1997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6578.0, "completions/max_terminated_length": 6578.0, "completions/mean_length": 5058.0, "completions/mean_terminated_length": 5058.0, "completions/min_length": 3538.0, "completions/min_terminated_length": 3538.0, "epoch": 0.04955971722683865, "grad_norm": 0.0, "learning_rate": 1.5e-09, "loss": 0.0, "num_tokens": 10998785.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 712.0, "completions/max_terminated_length": 712.0, "completions/mean_length": 578.5, "completions/mean_terminated_length": 578.5, "completions/min_length": 445.0, "completions/min_terminated_length": 445.0, "epoch": 0.04958452189011534, "grad_norm": 0.0, "learning_rate": 1e-09, "loss": 0.0, "num_tokens": 11000796.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 1999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 6205.0, "completions/max_terminated_length": 6205.0, "completions/mean_length": 6118.0, "completions/mean_terminated_length": 6118.0, "completions/min_length": 6031.0, "completions/min_terminated_length": 6031.0, "epoch": 0.049609326553392036, "grad_norm": 0.0, "learning_rate": 5e-10, "loss": 0.0, "num_tokens": 11013958.0, "reward": 1.0, "reward_std": 0.0, "rewards/accuracy_reward/mean": 1.0, "rewards/accuracy_reward/std": 0.0, "step": 2000 }, { "epoch": 0.049609326553392036, "step": 2000, "total_flos": 0.0, "train_loss": -0.07443550048445467, "train_runtime": 53119.2066, "train_samples_per_second": 0.075, "train_steps_per_second": 0.038 } ], "logging_steps": 1, "max_steps": 2000, "num_input_tokens_seen": 11013958, "num_train_epochs": 1, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }