{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.3408098494046478, "eval_steps": 500, "global_step": 2000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1623.38671875, "completions/mean_terminated_length": 1094.482421875, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.32960883900523186, "epoch": 0.00017040492470232388, "frac_reward_zero_std": 0.5, "grad_norm": 0.11354459077119827, "learning_rate": 1e-06, "loss": 0.0487, "num_tokens": 494835.0, "reward": 0.21875, "reward_std": 0.19970625638961792, "rewards/simpleverify_reward/mean": 0.21875, "rewards/simpleverify_reward/std": 0.41420844197273254, "step": 1, "tools/generated_tokens": 6151.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.68359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1757.83203125, "completions/mean_terminated_length": 1130.9259033203125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.3407264407724142, "epoch": 0.00034080984940464777, "frac_reward_zero_std": 0.5, "grad_norm": 0.12227898836135864, "learning_rate": 1e-06, "loss": 0.0191, "num_tokens": 1032424.0, "reward": 0.21875, "reward_std": 0.17978152632713318, "rewards/simpleverify_reward/mean": 0.21875, "rewards/simpleverify_reward/std": 0.41420844197273254, "step": 2, "tools/generated_tokens": 6749.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2013.0, "completions/mean_length": 1809.6796875, "completions/mean_terminated_length": 1064.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.3413053434342146, "epoch": 0.0005112147741069717, "frac_reward_zero_std": 0.6875, "grad_norm": 0.11295190453529358, "learning_rate": 1e-06, "loss": 0.0202, "num_tokens": 1591990.0, "reward": 0.14453125, "reward_std": 0.1115587055683136, "rewards/simpleverify_reward/mean": 0.14453125, "rewards/simpleverify_reward/std": 0.35231640934944153, "step": 3, "tools/generated_tokens": 7393.6875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.7265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.56640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1745.20703125, "completions/mean_terminated_length": 1349.7117919921875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2993578836321831, "epoch": 0.0006816196988092955, "frac_reward_zero_std": 0.5, "grad_norm": 0.11197676509618759, "learning_rate": 1e-06, "loss": 0.0245, "num_tokens": 2119995.0, "reward": 0.3125, "reward_std": 0.17499089241027832, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 4, "tools/generated_tokens": 6441.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.37890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1406.22265625, "completions/mean_terminated_length": 1014.704345703125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.3491258881986141, "epoch": 0.0008520246235116195, "frac_reward_zero_std": 0.3125, "grad_norm": 0.14115206897258759, "learning_rate": 1e-06, "loss": 0.0391, "num_tokens": 2563172.0, "reward": 0.33984375, "reward_std": 0.2401386797428131, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 5, "tools/generated_tokens": 5398.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.94921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.56640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1625.42578125, "completions/mean_terminated_length": 1073.4324951171875, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.45098429545760155, "epoch": 0.0010224295482139435, "frac_reward_zero_std": 0.375, "grad_norm": 0.15496528148651123, "learning_rate": 1e-06, "loss": 0.0365, "num_tokens": 3068193.0, "reward": 0.2734375, "reward_std": 0.2371043860912323, "rewards/simpleverify_reward/mean": 0.2734375, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 6, "tools/generated_tokens": 6705.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1647.97265625, "completions/mean_terminated_length": 1215.4227294921875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.3355453088879585, "epoch": 0.0011928344729162674, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15678000450134277, "learning_rate": 1e-06, "loss": 0.0481, "num_tokens": 3577610.0, "reward": 0.36328125, "reward_std": 0.25734785199165344, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 7, "tools/generated_tokens": 6591.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1603.62109375, "completions/mean_terminated_length": 1092.0252685546875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.35666714422404766, "epoch": 0.001363239397618591, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12019651383161545, "learning_rate": 1e-06, "loss": 0.002, "num_tokens": 4080745.0, "reward": 0.17578125, "reward_std": 0.17626741528511047, "rewards/simpleverify_reward/mean": 0.17578125, "rewards/simpleverify_reward/std": 0.3813795745372772, "step": 8, "tools/generated_tokens": 6115.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.57421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1702.17578125, "completions/mean_terminated_length": 1235.8072509765625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.36048993095755577, "epoch": 0.001533644322320915, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11520885676145554, "learning_rate": 1e-06, "loss": 0.021, "num_tokens": 4598038.0, "reward": 0.15234375, "reward_std": 0.17712010443210602, "rewards/simpleverify_reward/mean": 0.15234375, "rewards/simpleverify_reward/std": 0.3600577116012573, "step": 9, "tools/generated_tokens": 6598.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.62109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1667.6640625, "completions/mean_terminated_length": 1044.2474365234375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.35148809291422367, "epoch": 0.001704049247023239, "frac_reward_zero_std": 0.375, "grad_norm": 0.13783320784568787, "learning_rate": 1e-06, "loss": 0.0335, "num_tokens": 5109184.0, "reward": 0.29296875, "reward_std": 0.2421935796737671, "rewards/simpleverify_reward/mean": 0.29296875, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 10, "tools/generated_tokens": 6475.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1735.77734375, "completions/mean_terminated_length": 1264.411865234375, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.3556172177195549, "epoch": 0.0018744541717255628, "frac_reward_zero_std": 0.6875, "grad_norm": 0.09855224937200546, "learning_rate": 1e-06, "loss": 0.0304, "num_tokens": 5643399.0, "reward": 0.203125, "reward_std": 0.10519562661647797, "rewards/simpleverify_reward/mean": 0.203125, "rewards/simpleverify_reward/std": 0.40311288833618164, "step": 11, "tools/generated_tokens": 6807.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.44140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1592.29296875, "completions/mean_terminated_length": 1232.19580078125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.32066681049764156, "epoch": 0.002044859096427887, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13096962869167328, "learning_rate": 1e-06, "loss": 0.0284, "num_tokens": 6133570.0, "reward": 0.3125, "reward_std": 0.17385752499103546, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 12, "tools/generated_tokens": 5880.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1815.23828125, "completions/mean_terminated_length": 1054.8834228515625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.3145635910332203, "epoch": 0.0022152640211302106, "frac_reward_zero_std": 0.6875, "grad_norm": 0.11477241665124893, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 6688703.0, "reward": 0.15625, "reward_std": 0.1354367733001709, "rewards/simpleverify_reward/mean": 0.15625, "rewards/simpleverify_reward/std": 0.3638034462928772, "step": 13, "tools/generated_tokens": 7303.23828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1666.7109375, "completions/mean_terminated_length": 1160.654541015625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.3039297014474869, "epoch": 0.0023856689458325348, "frac_reward_zero_std": 0.5, "grad_norm": 0.13038323819637299, "learning_rate": 1e-06, "loss": 0.0159, "num_tokens": 7212133.0, "reward": 0.22265625, "reward_std": 0.18738040328025818, "rewards/simpleverify_reward/mean": 0.22265625, "rewards/simpleverify_reward/std": 0.41684433817863464, "step": 14, "tools/generated_tokens": 6506.72265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1575.6953125, "completions/mean_terminated_length": 1171.847900390625, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.40485683642327785, "epoch": 0.0025560738705348585, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15688402950763702, "learning_rate": 1e-06, "loss": 0.0193, "num_tokens": 7701367.0, "reward": 0.30859375, "reward_std": 0.2559266686439514, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 15, "tools/generated_tokens": 6167.703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.2421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1618.70703125, "completions/mean_terminated_length": 1124.4874267578125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.35048897564411163, "epoch": 0.002726478795237182, "frac_reward_zero_std": 0.5, "grad_norm": 0.17330823838710785, "learning_rate": 1e-06, "loss": 0.0364, "num_tokens": 8201052.0, "reward": 0.3671875, "reward_std": 0.15261822938919067, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 16, "tools/generated_tokens": 6266.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1644.9765625, "completions/mean_terminated_length": 1215.9595947265625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.36185348220169544, "epoch": 0.0028968837199395063, "frac_reward_zero_std": 0.5, "grad_norm": 0.11988542228937149, "learning_rate": 1e-06, "loss": 0.0412, "num_tokens": 8714230.0, "reward": 0.3203125, "reward_std": 0.1737399697303772, "rewards/simpleverify_reward/mean": 0.3203125, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 17, "tools/generated_tokens": 6420.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.47265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1607.0625, "completions/mean_terminated_length": 1211.86669921875, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.3239676281809807, "epoch": 0.00306728864464183, "frac_reward_zero_std": 0.5, "grad_norm": 0.11701221764087677, "learning_rate": 1e-06, "loss": 0.0026, "num_tokens": 9216486.0, "reward": 0.390625, "reward_std": 0.1892854869365692, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 18, "tools/generated_tokens": 6311.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.48828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1551.7265625, "completions/mean_terminated_length": 1078.198486328125, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.3650179672986269, "epoch": 0.003237693569344154, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12444363534450531, "learning_rate": 1e-06, "loss": 0.0311, "num_tokens": 9699552.0, "reward": 0.35546875, "reward_std": 0.16483555734157562, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 19, "tools/generated_tokens": 5847.74609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1714.6875, "completions/mean_terminated_length": 1194.72998046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.35892440751194954, "epoch": 0.003408098494046478, "frac_reward_zero_std": 0.625, "grad_norm": 0.12233246117830276, "learning_rate": 1e-06, "loss": 0.0165, "num_tokens": 10234128.0, "reward": 0.1953125, "reward_std": 0.15176509320735931, "rewards/simpleverify_reward/mean": 0.1953125, "rewards/simpleverify_reward/std": 0.39721766114234924, "step": 20, "tools/generated_tokens": 7010.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1769.3046875, "completions/mean_terminated_length": 1096.719970703125, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.3672831766307354, "epoch": 0.003578503418748802, "frac_reward_zero_std": 0.625, "grad_norm": 0.09308706223964691, "learning_rate": 1e-06, "loss": 0.017, "num_tokens": 10780926.0, "reward": 0.14453125, "reward_std": 0.138350710272789, "rewards/simpleverify_reward/mean": 0.14453125, "rewards/simpleverify_reward/std": 0.35231640934944153, "step": 21, "tools/generated_tokens": 7305.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1680.05078125, "completions/mean_terminated_length": 1191.727294921875, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.2959140334278345, "epoch": 0.0037489083434511256, "frac_reward_zero_std": 0.5, "grad_norm": 0.10992983728647232, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 11303227.0, "reward": 0.28125, "reward_std": 0.1999323070049286, "rewards/simpleverify_reward/mean": 0.28125, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 22, "tools/generated_tokens": 6648.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.42578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.52734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1579.7734375, "completions/mean_terminated_length": 1057.4296875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.3190094195306301, "epoch": 0.00391931326815345, "frac_reward_zero_std": 0.5, "grad_norm": 0.1401917040348053, "learning_rate": 1e-06, "loss": 0.0251, "num_tokens": 11800945.0, "reward": 0.33984375, "reward_std": 0.1824694126844406, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 23, "tools/generated_tokens": 6131.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.58984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1689.01953125, "completions/mean_terminated_length": 1172.79052734375, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.35156455263495445, "epoch": 0.004089718192855774, "frac_reward_zero_std": 0.375, "grad_norm": 0.19344040751457214, "learning_rate": 1e-06, "loss": 0.031, "num_tokens": 12317702.0, "reward": 0.2890625, "reward_std": 0.20389671623706818, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 24, "tools/generated_tokens": 6561.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1676.98046875, "completions/mean_terminated_length": 1116.813720703125, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.341730497777462, "epoch": 0.004260123117558097, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11004383862018585, "learning_rate": 1e-06, "loss": 0.0323, "num_tokens": 12837921.0, "reward": 0.1875, "reward_std": 0.18843428790569305, "rewards/simpleverify_reward/mean": 0.1875, "rewards/simpleverify_reward/std": 0.3910769522190094, "step": 25, "tools/generated_tokens": 6604.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1544.796875, "completions/mean_terminated_length": 1107.72265625, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "entropy": 0.31659174151718616, "epoch": 0.004430528042260421, "frac_reward_zero_std": 0.5625, "grad_norm": 0.10992801189422607, "learning_rate": 1e-06, "loss": 0.021, "num_tokens": 13324413.0, "reward": 0.2890625, "reward_std": 0.1932636797428131, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 26, "tools/generated_tokens": 6064.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.38671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1475.1015625, "completions/mean_terminated_length": 1113.85986328125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.31422682851552963, "epoch": 0.004600932966962745, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1931154876947403, "learning_rate": 1e-06, "loss": 0.0217, "num_tokens": 13792375.0, "reward": 0.46875, "reward_std": 0.2512108087539673, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 27, "tools/generated_tokens": 5867.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2004.0, "completions/mean_length": 1601.94921875, "completions/mean_terminated_length": 1134.488037109375, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.32225533202290535, "epoch": 0.0047713378916650695, "frac_reward_zero_std": 0.4375, "grad_norm": 0.12505559623241425, "learning_rate": 1e-06, "loss": 0.0454, "num_tokens": 14291850.0, "reward": 0.4140625, "reward_std": 0.21039125323295593, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 28, "tools/generated_tokens": 6393.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.55078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1672.54296875, "completions/mean_terminated_length": 1212.2086181640625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.32055498845875263, "epoch": 0.004941742816367393, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16482892632484436, "learning_rate": 1e-06, "loss": 0.0237, "num_tokens": 14802213.0, "reward": 0.375, "reward_std": 0.24836406111717224, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 29, "tools/generated_tokens": 6160.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1507.3359375, "completions/mean_terminated_length": 1059.3642578125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.36279159784317017, "epoch": 0.005112147741069717, "frac_reward_zero_std": 0.5, "grad_norm": 0.12792861461639404, "learning_rate": 1e-06, "loss": 0.0252, "num_tokens": 15271867.0, "reward": 0.34765625, "reward_std": 0.21448004245758057, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 30, "tools/generated_tokens": 5795.3359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.43359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1525.83984375, "completions/mean_terminated_length": 1126.1171875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.40453204698860645, "epoch": 0.005282552665772041, "frac_reward_zero_std": 0.5, "grad_norm": 0.14503608644008636, "learning_rate": 1e-06, "loss": 0.0195, "num_tokens": 15751746.0, "reward": 0.3046875, "reward_std": 0.2298790067434311, "rewards/simpleverify_reward/mean": 0.3046875, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 31, "tools/generated_tokens": 6157.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1549.65234375, "completions/mean_terminated_length": 1109.9559326171875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.3383567910641432, "epoch": 0.005452957590474364, "frac_reward_zero_std": 0.375, "grad_norm": 0.1525263786315918, "learning_rate": 1e-06, "loss": 0.0392, "num_tokens": 16243353.0, "reward": 0.38671875, "reward_std": 0.2071847766637802, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 32, "tools/generated_tokens": 5877.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2004.0, "completions/mean_length": 1602.7265625, "completions/mean_terminated_length": 1136.112060546875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.36383174173533916, "epoch": 0.0056233625151766884, "frac_reward_zero_std": 0.25, "grad_norm": 0.15686126053333282, "learning_rate": 1e-06, "loss": 0.024, "num_tokens": 16742019.0, "reward": 0.3359375, "reward_std": 0.2700601816177368, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 33, "tools/generated_tokens": 6618.74609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1344.25390625, "completions/mean_terminated_length": 988.2470703125, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.36758890748023987, "epoch": 0.005793767439879013, "frac_reward_zero_std": 0.5, "grad_norm": 0.16702856123447418, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 17168116.0, "reward": 0.28125, "reward_std": 0.22981059551239014, "rewards/simpleverify_reward/mean": 0.28125, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 34, "tools/generated_tokens": 5288.26171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1457.4375, "completions/mean_terminated_length": 1137.2650146484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.3349355608224869, "epoch": 0.005964172364581337, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11891477555036545, "learning_rate": 1e-06, "loss": 0.03, "num_tokens": 17627588.0, "reward": 0.3359375, "reward_std": 0.17693254351615906, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 35, "tools/generated_tokens": 5873.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.15625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1471.2421875, "completions/mean_terminated_length": 1022.6597290039062, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.3687310889363289, "epoch": 0.00613457728928366, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15980218350887299, "learning_rate": 1e-06, "loss": 0.0302, "num_tokens": 18087282.0, "reward": 0.3671875, "reward_std": 0.3007515072822571, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 36, "tools/generated_tokens": 5999.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.44921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1500.85546875, "completions/mean_terminated_length": 1054.616943359375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.3143069688230753, "epoch": 0.006304982213985984, "frac_reward_zero_std": 0.5, "grad_norm": 0.12282077223062515, "learning_rate": 1e-06, "loss": 0.0125, "num_tokens": 18553053.0, "reward": 0.2734375, "reward_std": 0.17862266302108765, "rewards/simpleverify_reward/mean": 0.2734375, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 37, "tools/generated_tokens": 5604.875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.00390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1652.72265625, "completions/mean_terminated_length": 1225.333251953125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.3574356138706207, "epoch": 0.006475387138688308, "frac_reward_zero_std": 0.625, "grad_norm": 0.08517470210790634, "learning_rate": 1e-06, "loss": 0.0257, "num_tokens": 19063382.0, "reward": 0.203125, "reward_std": 0.14704003930091858, "rewards/simpleverify_reward/mean": 0.203125, "rewards/simpleverify_reward/std": 0.40311288833618164, "step": 38, "tools/generated_tokens": 6236.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 1862.078125, "completions/mean_terminated_length": 1254.7333984375, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.32641259767115116, "epoch": 0.006645792063390632, "frac_reward_zero_std": 0.8125, "grad_norm": 0.05535029247403145, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 19634682.0, "reward": 0.0703125, "reward_std": 0.08539125323295593, "rewards/simpleverify_reward/mean": 0.0703125, "rewards/simpleverify_reward/std": 0.2561737895011902, "step": 39, "tools/generated_tokens": 7454.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1432.1953125, "completions/mean_terminated_length": 1050.253173828125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.38021427020430565, "epoch": 0.006816196988092956, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13984593749046326, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 20088108.0, "reward": 0.53515625, "reward_std": 0.18213960528373718, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 40, "tools/generated_tokens": 5464.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.96875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 1978.0, "completions/mean_length": 1794.15625, "completions/mean_terminated_length": 1092.36767578125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.3053403776139021, "epoch": 0.00698660191279528, "frac_reward_zero_std": 0.4375, "grad_norm": 0.11249354481697083, "learning_rate": 1e-06, "loss": 0.0218, "num_tokens": 20640228.0, "reward": 0.2578125, "reward_std": 0.1900683045387268, "rewards/simpleverify_reward/mean": 0.2578125, "rewards/simpleverify_reward/std": 0.4382871091365814, "step": 41, "tools/generated_tokens": 7122.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1605.3515625, "completions/mean_terminated_length": 1095.75634765625, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.3565551396459341, "epoch": 0.007157006837497604, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14736947417259216, "learning_rate": 1e-06, "loss": 0.0609, "num_tokens": 21142254.0, "reward": 0.25390625, "reward_std": 0.17968884110450745, "rewards/simpleverify_reward/mean": 0.25390625, "rewards/simpleverify_reward/std": 0.4360972046852112, "step": 42, "tools/generated_tokens": 6397.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1620.98046875, "completions/mean_terminated_length": 1180.40478515625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.31288580037653446, "epoch": 0.007327411762199928, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1469164937734604, "learning_rate": 1e-06, "loss": 0.0412, "num_tokens": 21642633.0, "reward": 0.35546875, "reward_std": 0.27475816011428833, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 43, "tools/generated_tokens": 6460.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1577.40234375, "completions/mean_terminated_length": 1187.4786376953125, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.37273502349853516, "epoch": 0.007497816686902251, "frac_reward_zero_std": 0.375, "grad_norm": 0.14188292622566223, "learning_rate": 1e-06, "loss": 0.0473, "num_tokens": 22148912.0, "reward": 0.30078125, "reward_std": 0.24556957185268402, "rewards/simpleverify_reward/mean": 0.30078125, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 44, "tools/generated_tokens": 6313.40625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1561.4453125, "completions/mean_terminated_length": 1145.4130859375, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.3211830984801054, "epoch": 0.007668221611604575, "frac_reward_zero_std": 0.5, "grad_norm": 0.11664831638336182, "learning_rate": 1e-06, "loss": 0.0315, "num_tokens": 22635714.0, "reward": 0.3671875, "reward_std": 0.2108054757118225, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 45, "tools/generated_tokens": 6201.46875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1571.8046875, "completions/mean_terminated_length": 1201.4722900390625, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.3283666502684355, "epoch": 0.0078386265363069, "frac_reward_zero_std": 0.5, "grad_norm": 0.139494389295578, "learning_rate": 1e-06, "loss": 0.0478, "num_tokens": 23124672.0, "reward": 0.41015625, "reward_std": 0.20382197201251984, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 46, "tools/generated_tokens": 5931.8515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1610.671875, "completions/mean_terminated_length": 1107.2017822265625, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.3429228141903877, "epoch": 0.008009031461009224, "frac_reward_zero_std": 0.6875, "grad_norm": 0.1430044174194336, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 23628716.0, "reward": 0.390625, "reward_std": 0.10331955552101135, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 47, "tools/generated_tokens": 6210.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.43359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1567.9765625, "completions/mean_terminated_length": 1200.5103759765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.35710458643734455, "epoch": 0.008179436385711548, "frac_reward_zero_std": 0.6875, "grad_norm": 0.12251273542642593, "learning_rate": 1e-06, "loss": 0.0403, "num_tokens": 24115142.0, "reward": 0.43359375, "reward_std": 0.12082535773515701, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 48, "tools/generated_tokens": 5895.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1555.0625, "completions/mean_terminated_length": 1195.3514404296875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.3559937682002783, "epoch": 0.00834984131041387, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1743675172328949, "learning_rate": 1e-06, "loss": 0.0072, "num_tokens": 24594118.0, "reward": 0.34765625, "reward_std": 0.17286168038845062, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 49, "tools/generated_tokens": 5531.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.94140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1506.8359375, "completions/mean_terminated_length": 1233.070556640625, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.31781978718936443, "epoch": 0.008520246235116194, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19409367442131042, "learning_rate": 1e-06, "loss": -0.0134, "num_tokens": 25071788.0, "reward": 0.4765625, "reward_std": 0.26829975843429565, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 50, "tools/generated_tokens": 5642.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1465.50390625, "completions/mean_terminated_length": 1160.3988037109375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.3548562824726105, "epoch": 0.008690651159818518, "frac_reward_zero_std": 0.375, "grad_norm": 0.1465141773223877, "learning_rate": 1e-06, "loss": 0.0526, "num_tokens": 25534637.0, "reward": 0.55859375, "reward_std": 0.2823812961578369, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 51, "tools/generated_tokens": 5801.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1404.16796875, "completions/mean_terminated_length": 1042.993896484375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.32291222736239433, "epoch": 0.008861056084520843, "frac_reward_zero_std": 0.5, "grad_norm": 0.1312384307384491, "learning_rate": 1e-06, "loss": 0.0331, "num_tokens": 25980520.0, "reward": 0.37890625, "reward_std": 0.23091968894004822, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 52, "tools/generated_tokens": 5452.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.37890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1470.67578125, "completions/mean_terminated_length": 1118.4716796875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.3883262947201729, "epoch": 0.009031461009223167, "frac_reward_zero_std": 0.5, "grad_norm": 0.1563533991575241, "learning_rate": 1e-06, "loss": -0.0133, "num_tokens": 26438693.0, "reward": 0.42578125, "reward_std": 0.18032719194889069, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 53, "tools/generated_tokens": 5510.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 1993.0, "completions/mean_length": 1653.49609375, "completions/mean_terminated_length": 1095.2452392578125, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.3353212848305702, "epoch": 0.00920186593392549, "frac_reward_zero_std": 0.625, "grad_norm": 0.09876301139593124, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 26953028.0, "reward": 0.33984375, "reward_std": 0.12082062661647797, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 54, "tools/generated_tokens": 6589.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.41015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1426.5625, "completions/mean_terminated_length": 1041.1138916015625, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.3311825506389141, "epoch": 0.009372270858627815, "frac_reward_zero_std": 0.5, "grad_norm": 0.14525867998600006, "learning_rate": 1e-06, "loss": 0.0471, "num_tokens": 27426644.0, "reward": 0.40625, "reward_std": 0.18923160433769226, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 55, "tools/generated_tokens": 5778.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1590.58203125, "completions/mean_terminated_length": 1111.216064453125, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.3789573274552822, "epoch": 0.009542675783330139, "frac_reward_zero_std": 0.375, "grad_norm": 0.15582266449928284, "learning_rate": 1e-06, "loss": 0.013, "num_tokens": 27919465.0, "reward": 0.296875, "reward_std": 0.250201940536499, "rewards/simpleverify_reward/mean": 0.296875, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 56, "tools/generated_tokens": 6278.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.42578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2005.0, "completions/mean_length": 1474.25, "completions/mean_terminated_length": 1048.8231201171875, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.3115955535322428, "epoch": 0.009713080708032461, "frac_reward_zero_std": 0.625, "grad_norm": 0.11107443273067474, "learning_rate": 1e-06, "loss": 0.0304, "num_tokens": 28390505.0, "reward": 0.4453125, "reward_std": 0.1519911289215088, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 57, "tools/generated_tokens": 5698.2578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2006.0, "completions/mean_length": 1476.76953125, "completions/mean_terminated_length": 1085.9276123046875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.3250275757163763, "epoch": 0.009883485632734786, "frac_reward_zero_std": 0.4375, "grad_norm": 0.12432430684566498, "learning_rate": 1e-06, "loss": -0.0352, "num_tokens": 28856958.0, "reward": 0.44921875, "reward_std": 0.19398343563079834, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 58, "tools/generated_tokens": 5876.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.41015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1483.8125, "completions/mean_terminated_length": 1091.4967041015625, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.32753048464655876, "epoch": 0.01005389055743711, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15509773790836334, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 29323134.0, "reward": 0.25390625, "reward_std": 0.22962586581707, "rewards/simpleverify_reward/mean": 0.25390625, "rewards/simpleverify_reward/std": 0.4360972046852112, "step": 59, "tools/generated_tokens": 5899.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.15625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.48828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1534.98828125, "completions/mean_terminated_length": 1045.488525390625, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.3313278928399086, "epoch": 0.010224295482139434, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15667590498924255, "learning_rate": 1e-06, "loss": -0.0044, "num_tokens": 29806059.0, "reward": 0.30859375, "reward_std": 0.17838552594184875, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 60, "tools/generated_tokens": 5863.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.61328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1671.04296875, "completions/mean_terminated_length": 1073.3837890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.3573396895080805, "epoch": 0.010394700406841758, "frac_reward_zero_std": 0.5, "grad_norm": 0.8728443384170532, "learning_rate": 1e-06, "loss": 0.0411, "num_tokens": 30323014.0, "reward": 0.26953125, "reward_std": 0.20187556743621826, "rewards/simpleverify_reward/mean": 0.26953125, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 61, "tools/generated_tokens": 6735.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.54296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1607.87890625, "completions/mean_terminated_length": 1085.01708984375, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.346057066693902, "epoch": 0.010565105331544082, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1445242166519165, "learning_rate": 1e-06, "loss": 0.0588, "num_tokens": 30833687.0, "reward": 0.38671875, "reward_std": 0.2286548763513565, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 62, "tools/generated_tokens": 6671.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2011.0, "completions/mean_length": 1459.421875, "completions/mean_terminated_length": 1094.3544921875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.3540453128516674, "epoch": 0.010735510256246406, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11674729734659195, "learning_rate": 1e-06, "loss": -0.0057, "num_tokens": 31292387.0, "reward": 0.4921875, "reward_std": 0.15074022114276886, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 63, "tools/generated_tokens": 5411.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.50390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1577.82421875, "completions/mean_terminated_length": 1100.251953125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.3563331104815006, "epoch": 0.010905915180948729, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13753943145275116, "learning_rate": 1e-06, "loss": 0.0228, "num_tokens": 31787462.0, "reward": 0.29296875, "reward_std": 0.25158432126045227, "rewards/simpleverify_reward/mean": 0.29296875, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 64, "tools/generated_tokens": 6313.83203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1578.86328125, "completions/mean_terminated_length": 1151.753662109375, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.3383399248123169, "epoch": 0.011076320105651053, "frac_reward_zero_std": 0.625, "grad_norm": 0.1068626344203949, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 32280979.0, "reward": 0.39453125, "reward_std": 0.13039018213748932, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 65, "tools/generated_tokens": 5890.87890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1475.69921875, "completions/mean_terminated_length": 1108.8590087890625, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.3287957701832056, "epoch": 0.011246725030353377, "frac_reward_zero_std": 0.5, "grad_norm": 0.1338070183992386, "learning_rate": 1e-06, "loss": 0.0038, "num_tokens": 32748454.0, "reward": 0.2890625, "reward_std": 0.2060832381248474, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 66, "tools/generated_tokens": 5475.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1533.21875, "completions/mean_terminated_length": 1157.567626953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.3912510294467211, "epoch": 0.011417129955055701, "frac_reward_zero_std": 0.625, "grad_norm": 0.18103700876235962, "learning_rate": 1e-06, "loss": 0.0275, "num_tokens": 33230286.0, "reward": 0.33984375, "reward_std": 0.1544148027896881, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 67, "tools/generated_tokens": 5981.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1698.58984375, "completions/mean_terminated_length": 1263.359619140625, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.3388095647096634, "epoch": 0.011587534879758025, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1136864721775055, "learning_rate": 1e-06, "loss": 0.037, "num_tokens": 33749509.0, "reward": 0.265625, "reward_std": 0.1779082715511322, "rewards/simpleverify_reward/mean": 0.265625, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 68, "tools/generated_tokens": 6474.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.66015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1711.87890625, "completions/mean_terminated_length": 1058.9654541015625, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.3804211299866438, "epoch": 0.01175793980446035, "frac_reward_zero_std": 0.625, "grad_norm": 0.10970202088356018, "learning_rate": 1e-06, "loss": 0.0261, "num_tokens": 34278182.0, "reward": 0.15625, "reward_std": 0.1355944126844406, "rewards/simpleverify_reward/mean": 0.15625, "rewards/simpleverify_reward/std": 0.3638034462928772, "step": 69, "tools/generated_tokens": 7031.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.47265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1503.2578125, "completions/mean_terminated_length": 1015.0147705078125, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.3382138181477785, "epoch": 0.011928344729162673, "frac_reward_zero_std": 0.375, "grad_norm": 0.15408317744731903, "learning_rate": 1e-06, "loss": 0.0628, "num_tokens": 34751448.0, "reward": 0.23828125, "reward_std": 0.2356673926115036, "rewards/simpleverify_reward/mean": 0.23828125, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 70, "tools/generated_tokens": 6231.265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.61328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1681.09375, "completions/mean_terminated_length": 1099.2322998046875, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.3789903335273266, "epoch": 0.012098749653864998, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1188197210431099, "learning_rate": 1e-06, "loss": 0.0079, "num_tokens": 35273824.0, "reward": 0.28125, "reward_std": 0.1504095196723938, "rewards/simpleverify_reward/mean": 0.28125, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 71, "tools/generated_tokens": 6985.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1689.72265625, "completions/mean_terminated_length": 1182.745361328125, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.3729398362338543, "epoch": 0.01226915457856732, "frac_reward_zero_std": 0.5, "grad_norm": 0.15884780883789062, "learning_rate": 1e-06, "loss": 0.0191, "num_tokens": 35798889.0, "reward": 0.328125, "reward_std": 0.20214100182056427, "rewards/simpleverify_reward/mean": 0.328125, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 72, "tools/generated_tokens": 6577.73828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.32421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1412.16796875, "completions/mean_terminated_length": 1107.121337890625, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.34434461034834385, "epoch": 0.012439559503269644, "frac_reward_zero_std": 0.5, "grad_norm": 0.1620829850435257, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 36248548.0, "reward": 0.43359375, "reward_std": 0.19619880616664886, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 73, "tools/generated_tokens": 5764.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1368.33984375, "completions/mean_terminated_length": 1076.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.3357909843325615, "epoch": 0.012609964427971968, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14101873338222504, "learning_rate": 1e-06, "loss": 0.0508, "num_tokens": 36680971.0, "reward": 0.5703125, "reward_std": 0.2199607938528061, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 74, "tools/generated_tokens": 5272.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.90625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.61328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1736.640625, "completions/mean_terminated_length": 1242.86865234375, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.35474786534905434, "epoch": 0.012780369352674292, "frac_reward_zero_std": 0.5625, "grad_norm": 0.10660536587238312, "learning_rate": 1e-06, "loss": 0.0109, "num_tokens": 37218543.0, "reward": 0.2890625, "reward_std": 0.16691282391548157, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 75, "tools/generated_tokens": 6608.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1656.92578125, "completions/mean_terminated_length": 1199.5762939453125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.3315989449620247, "epoch": 0.012950774277376616, "frac_reward_zero_std": 0.75, "grad_norm": 0.1214306652545929, "learning_rate": 1e-06, "loss": 0.0243, "num_tokens": 37729004.0, "reward": 0.2578125, "reward_std": 0.10065875202417374, "rewards/simpleverify_reward/mean": 0.2578125, "rewards/simpleverify_reward/std": 0.4382871091365814, "step": 76, "tools/generated_tokens": 6480.953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1479.21875, "completions/mean_terminated_length": 1064.1689453125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.3326445445418358, "epoch": 0.01312117920207894, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1574394851922989, "learning_rate": 1e-06, "loss": 0.0415, "num_tokens": 38193348.0, "reward": 0.4296875, "reward_std": 0.258681058883667, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 77, "tools/generated_tokens": 5703.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.32421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1442.6796875, "completions/mean_terminated_length": 1152.2716064453125, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.342384722083807, "epoch": 0.013291584126781265, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15261264145374298, "learning_rate": 1e-06, "loss": -0.0514, "num_tokens": 38640738.0, "reward": 0.5390625, "reward_std": 0.23635752499103546, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 78, "tools/generated_tokens": 5114.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.79296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.37890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1415.203125, "completions/mean_terminated_length": 1029.1572265625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.3426816575229168, "epoch": 0.013461989051483587, "frac_reward_zero_std": 0.1875, "grad_norm": 0.2051922231912613, "learning_rate": 1e-06, "loss": 0.0292, "num_tokens": 39090486.0, "reward": 0.47265625, "reward_std": 0.30645641684532166, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 79, "tools/generated_tokens": 5935.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1454.515625, "completions/mean_terminated_length": 1110.1605224609375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.33010238222777843, "epoch": 0.013632393976185911, "frac_reward_zero_std": 0.375, "grad_norm": 0.15772663056850433, "learning_rate": 1e-06, "loss": 0.0053, "num_tokens": 39554842.0, "reward": 0.47265625, "reward_std": 0.2092868983745575, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 80, "tools/generated_tokens": 5686.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1429.90234375, "completions/mean_terminated_length": 1033.6859130859375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.35302944108843803, "epoch": 0.013802798900888235, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15112242102622986, "learning_rate": 1e-06, "loss": 0.0263, "num_tokens": 40011553.0, "reward": 0.41015625, "reward_std": 0.28301119804382324, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 81, "tools/generated_tokens": 5821.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1536.80859375, "completions/mean_terminated_length": 1187.0460205078125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.376364478841424, "epoch": 0.01397320382559056, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17073360085487366, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 40497424.0, "reward": 0.32421875, "reward_std": 0.27595800161361694, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 82, "tools/generated_tokens": 6096.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.2265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.37109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1381.19140625, "completions/mean_terminated_length": 987.73291015625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.3292817212641239, "epoch": 0.014143608750292884, "frac_reward_zero_std": 0.25, "grad_norm": 0.5036759972572327, "learning_rate": 1e-06, "loss": 0.0452, "num_tokens": 40934545.0, "reward": 0.33203125, "reward_std": 0.2770320177078247, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 83, "tools/generated_tokens": 5365.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1399.3359375, "completions/mean_terminated_length": 1174.015869140625, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.33120965771377087, "epoch": 0.014314013674995208, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16501937806606293, "learning_rate": 1e-06, "loss": 0.0097, "num_tokens": 41375990.0, "reward": 0.5, "reward_std": 0.21643753349781036, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 84, "tools/generated_tokens": 5175.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1628.33203125, "completions/mean_terminated_length": 1071.318115234375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.3914438560605049, "epoch": 0.014484418599697532, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11321353167295456, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 41885115.0, "reward": 0.2734375, "reward_std": 0.1892854869365692, "rewards/simpleverify_reward/mean": 0.2734375, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 85, "tools/generated_tokens": 6524.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1390.75, "completions/mean_terminated_length": 1046.482177734375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.35446988977491856, "epoch": 0.014654823524399856, "frac_reward_zero_std": 0.5, "grad_norm": 0.1685134768486023, "learning_rate": 1e-06, "loss": 0.0177, "num_tokens": 42319931.0, "reward": 0.48046875, "reward_std": 0.17473775148391724, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 86, "tools/generated_tokens": 5270.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.89453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1650.60546875, "completions/mean_terminated_length": 1200.2333984375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.31209629215300083, "epoch": 0.014825228449102178, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16722668707370758, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 42831142.0, "reward": 0.33984375, "reward_std": 0.2480090707540512, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 87, "tools/generated_tokens": 6098.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.44140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1558.86328125, "completions/mean_terminated_length": 1172.3636474609375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.32257607765495777, "epoch": 0.014995633373804503, "frac_reward_zero_std": 0.5, "grad_norm": 0.262336403131485, "learning_rate": 1e-06, "loss": 0.024, "num_tokens": 43320419.0, "reward": 0.27734375, "reward_std": 0.1941438913345337, "rewards/simpleverify_reward/mean": 0.27734375, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 88, "tools/generated_tokens": 5806.87890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.07421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1568.30859375, "completions/mean_terminated_length": 1240.0986328125, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.31894766725599766, "epoch": 0.015166038298506827, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1430114060640335, "learning_rate": 1e-06, "loss": 0.0263, "num_tokens": 43804674.0, "reward": 0.33984375, "reward_std": 0.2531842887401581, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 89, "tools/generated_tokens": 6024.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.41015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1492.8359375, "completions/mean_terminated_length": 1106.8145751953125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.35931413620710373, "epoch": 0.01533644322320915, "frac_reward_zero_std": 0.5625, "grad_norm": 0.18770164251327515, "learning_rate": 1e-06, "loss": -0.0091, "num_tokens": 44273656.0, "reward": 0.42578125, "reward_std": 0.17781084775924683, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 90, "tools/generated_tokens": 5804.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1410.2578125, "completions/mean_terminated_length": 1135.9384765625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.3204840440303087, "epoch": 0.015506848147911475, "frac_reward_zero_std": 0.375, "grad_norm": 0.16799981892108917, "learning_rate": 1e-06, "loss": 0.0144, "num_tokens": 44713354.0, "reward": 0.41796875, "reward_std": 0.25784093141555786, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 91, "tools/generated_tokens": 5010.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.41796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1517.09765625, "completions/mean_terminated_length": 1135.852294921875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.3315076846629381, "epoch": 0.0156772530726138, "frac_reward_zero_std": 0.375, "grad_norm": 0.13466773927211761, "learning_rate": 1e-06, "loss": -0.0234, "num_tokens": 45191827.0, "reward": 0.27734375, "reward_std": 0.2296258509159088, "rewards/simpleverify_reward/mean": 0.27734375, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 92, "tools/generated_tokens": 6101.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1620.7421875, "completions/mean_terminated_length": 1151.4835205078125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.32753794454038143, "epoch": 0.015847657997316123, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11930279433727264, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 45696193.0, "reward": 0.3515625, "reward_std": 0.19025646150112152, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 93, "tools/generated_tokens": 6428.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1418.53125, "completions/mean_terminated_length": 1040.8499755859375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.336428202688694, "epoch": 0.016018062922018447, "frac_reward_zero_std": 0.375, "grad_norm": 0.16664044559001923, "learning_rate": 1e-06, "loss": 0.0255, "num_tokens": 46151977.0, "reward": 0.4296875, "reward_std": 0.27219003438949585, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 94, "tools/generated_tokens": 5842.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1369.5859375, "completions/mean_terminated_length": 1161.9132080078125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.29961889889091253, "epoch": 0.01618846784672077, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1664113849401474, "learning_rate": 1e-06, "loss": 0.0188, "num_tokens": 46583567.0, "reward": 0.45703125, "reward_std": 0.30801716446876526, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 95, "tools/generated_tokens": 5073.6015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.44140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1489.640625, "completions/mean_terminated_length": 1048.4405517578125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.31038382835686207, "epoch": 0.016358872771423096, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17741236090660095, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 47056003.0, "reward": 0.25390625, "reward_std": 0.2806849479675293, "rewards/simpleverify_reward/mean": 0.25390625, "rewards/simpleverify_reward/std": 0.4360972046852112, "step": 96, "tools/generated_tokens": 6121.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1624.23046875, "completions/mean_terminated_length": 1024.556640625, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.33994131349027157, "epoch": 0.01652927769612542, "frac_reward_zero_std": 0.75, "grad_norm": 0.16670498251914978, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 47556878.0, "reward": 0.3359375, "reward_std": 0.11022830009460449, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 97, "tools/generated_tokens": 6456.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1402.65625, "completions/mean_terminated_length": 1076.188232421875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.3517347723245621, "epoch": 0.01669968262082774, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14122840762138367, "learning_rate": 1e-06, "loss": 0.046, "num_tokens": 47995766.0, "reward": 0.4609375, "reward_std": 0.16515429317951202, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 98, "tools/generated_tokens": 5322.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1427.7109375, "completions/mean_terminated_length": 1010.1372680664062, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.3101299777626991, "epoch": 0.016870087545530064, "frac_reward_zero_std": 0.625, "grad_norm": 0.1458555907011032, "learning_rate": 1e-06, "loss": -0.0055, "num_tokens": 48452812.0, "reward": 0.46484375, "reward_std": 0.13466504216194153, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 99, "tools/generated_tokens": 5587.72265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.03125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1206.11328125, "completions/mean_terminated_length": 1006.8260498046875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.3743795230984688, "epoch": 0.01704049247023239, "frac_reward_zero_std": 0.375, "grad_norm": 0.16655708849430084, "learning_rate": 1e-06, "loss": 0.0213, "num_tokens": 48853001.0, "reward": 0.43359375, "reward_std": 0.24856583774089813, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 100, "tools/generated_tokens": 5054.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1231.859375, "completions/mean_terminated_length": 1048.3253173828125, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.3789853770285845, "epoch": 0.017210897394934713, "frac_reward_zero_std": 0.375, "grad_norm": 0.18441075086593628, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 49252037.0, "reward": 0.390625, "reward_std": 0.2340293675661087, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 101, "tools/generated_tokens": 4663.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.67578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1294.41015625, "completions/mean_terminated_length": 1010.8064575195312, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.3455806504935026, "epoch": 0.017381302319637037, "frac_reward_zero_std": 0.125, "grad_norm": 0.22286009788513184, "learning_rate": 1e-06, "loss": -0.0133, "num_tokens": 49678414.0, "reward": 0.4296875, "reward_std": 0.3573821485042572, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 102, "tools/generated_tokens": 5310.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.33984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1427.609375, "completions/mean_terminated_length": 1108.2366943359375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.3169189915060997, "epoch": 0.01755170724433936, "frac_reward_zero_std": 0.375, "grad_norm": 0.16372622549533844, "learning_rate": 1e-06, "loss": 0.0053, "num_tokens": 50135642.0, "reward": 0.484375, "reward_std": 0.2323840707540512, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 103, "tools/generated_tokens": 5515.625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1227.12890625, "completions/mean_terminated_length": 1092.8045654296875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.2742748726159334, "epoch": 0.017722112169041685, "frac_reward_zero_std": 0.25, "grad_norm": 0.17461593449115753, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 50531243.0, "reward": 0.60546875, "reward_std": 0.2772725820541382, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 104, "tools/generated_tokens": 4291.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1237.78515625, "completions/mean_terminated_length": 1021.1930541992188, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.3275550380349159, "epoch": 0.01789251709374401, "frac_reward_zero_std": 0.375, "grad_norm": 0.1666494458913803, "learning_rate": 1e-06, "loss": -0.0141, "num_tokens": 50934276.0, "reward": 0.3828125, "reward_std": 0.2369977980852127, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 105, "tools/generated_tokens": 4749.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1306.1484375, "completions/mean_terminated_length": 1004.5164794921875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.3629078324884176, "epoch": 0.018062922018446333, "frac_reward_zero_std": 0.3125, "grad_norm": 0.14647015929222107, "learning_rate": 1e-06, "loss": 0.0131, "num_tokens": 51352874.0, "reward": 0.51171875, "reward_std": 0.2669561505317688, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 106, "tools/generated_tokens": 5010.16015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.49609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1560.33984375, "completions/mean_terminated_length": 1080.2713623046875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.34390639141201973, "epoch": 0.018233326943148657, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14581336081027985, "learning_rate": 1e-06, "loss": 0.0334, "num_tokens": 51848881.0, "reward": 0.3203125, "reward_std": 0.22075963020324707, "rewards/simpleverify_reward/mean": 0.3203125, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 107, "tools/generated_tokens": 6000.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.16796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1402.94921875, "completions/mean_terminated_length": 1028.6605224609375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.35339405201375484, "epoch": 0.01840373186785098, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14013394713401794, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 52293972.0, "reward": 0.46484375, "reward_std": 0.17339344322681427, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 108, "tools/generated_tokens": 5098.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1309.03125, "completions/mean_terminated_length": 1002.8287963867188, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.31919316854327917, "epoch": 0.018574136792553306, "frac_reward_zero_std": 0.375, "grad_norm": 0.15078844130039215, "learning_rate": 1e-06, "loss": 0.0357, "num_tokens": 52722892.0, "reward": 0.3984375, "reward_std": 0.2612866461277008, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 109, "tools/generated_tokens": 5701.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.33984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1321.078125, "completions/mean_terminated_length": 946.8638916015625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.32734917663037777, "epoch": 0.01874454171725563, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18454772233963013, "learning_rate": 1e-06, "loss": 0.0383, "num_tokens": 53147040.0, "reward": 0.30078125, "reward_std": 0.23520077764987946, "rewards/simpleverify_reward/mean": 0.30078125, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 110, "tools/generated_tokens": 5673.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.44921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1435.4375, "completions/mean_terminated_length": 935.8368530273438, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.4320798348635435, "epoch": 0.018914946641957954, "frac_reward_zero_std": 0.5, "grad_norm": 0.1485069841146469, "learning_rate": 1e-06, "loss": 0.0603, "num_tokens": 53611456.0, "reward": 0.32421875, "reward_std": 0.1660325825214386, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 111, "tools/generated_tokens": 5859.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2019.0, "completions/mean_length": 1516.96875, "completions/mean_terminated_length": 1116.876708984375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.323065472766757, "epoch": 0.019085351566660278, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14810976386070251, "learning_rate": 1e-06, "loss": 0.023, "num_tokens": 54085768.0, "reward": 0.3046875, "reward_std": 0.19531384110450745, "rewards/simpleverify_reward/mean": 0.3046875, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 112, "tools/generated_tokens": 5524.96875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.43359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1487.953125, "completions/mean_terminated_length": 1059.248291015625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.35362469032406807, "epoch": 0.0192557564913626, "frac_reward_zero_std": 0.375, "grad_norm": 0.14924024045467377, "learning_rate": 1e-06, "loss": 0.0351, "num_tokens": 54561228.0, "reward": 0.30859375, "reward_std": 0.26145052909851074, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 113, "tools/generated_tokens": 6151.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1527.27734375, "completions/mean_terminated_length": 1171.006591796875, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.3045827057212591, "epoch": 0.019426161416064923, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15383483469486237, "learning_rate": 1e-06, "loss": 0.0278, "num_tokens": 55037155.0, "reward": 0.375, "reward_std": 0.2668628990650177, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 114, "tools/generated_tokens": 5895.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1442.7734375, "completions/mean_terminated_length": 1091.5926513671875, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.3237005192786455, "epoch": 0.019596566340767247, "frac_reward_zero_std": 0.25, "grad_norm": 0.14513492584228516, "learning_rate": 1e-06, "loss": 0.0432, "num_tokens": 55487081.0, "reward": 0.4296875, "reward_std": 0.28388863801956177, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 115, "tools/generated_tokens": 5282.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1450.76171875, "completions/mean_terminated_length": 1055.1883544921875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.3501081932336092, "epoch": 0.01976697126546957, "frac_reward_zero_std": 0.375, "grad_norm": 0.17928148806095123, "learning_rate": 1e-06, "loss": 0.0324, "num_tokens": 55944428.0, "reward": 0.40625, "reward_std": 0.27346593141555786, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 116, "tools/generated_tokens": 5874.76953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1071.96484375, "completions/mean_terminated_length": 1015.4999389648438, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.29012203868478537, "epoch": 0.019937376190171895, "frac_reward_zero_std": 0.4375, "grad_norm": 0.24156872928142548, "learning_rate": 1e-06, "loss": -0.0235, "num_tokens": 56293939.0, "reward": 0.5390625, "reward_std": 0.20067915320396423, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 117, "tools/generated_tokens": 3511.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1999.0, "completions/mean_length": 1561.28515625, "completions/mean_terminated_length": 1158.0072021484375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.33826029673218727, "epoch": 0.02010778111487422, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1409139335155487, "learning_rate": 1e-06, "loss": 0.0232, "num_tokens": 56775820.0, "reward": 0.30078125, "reward_std": 0.23041339218616486, "rewards/simpleverify_reward/mean": 0.30078125, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 118, "tools/generated_tokens": 6257.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1306.16015625, "completions/mean_terminated_length": 1130.5555419921875, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.312307920306921, "epoch": 0.020278186039576544, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14839047193527222, "learning_rate": 1e-06, "loss": -0.0031, "num_tokens": 57196725.0, "reward": 0.51953125, "reward_std": 0.2231852114200592, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 119, "tools/generated_tokens": 4626.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1988.0, "completions/mean_length": 1559.515625, "completions/mean_terminated_length": 1114.7835693359375, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.3601351138204336, "epoch": 0.020448590964278868, "frac_reward_zero_std": 0.375, "grad_norm": 0.16119074821472168, "learning_rate": 1e-06, "loss": 0.0125, "num_tokens": 57681305.0, "reward": 0.3828125, "reward_std": 0.24304214119911194, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 120, "tools/generated_tokens": 6255.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1400.7734375, "completions/mean_terminated_length": 1111.8983154296875, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.36240669898688793, "epoch": 0.020618995888981192, "frac_reward_zero_std": 0.625, "grad_norm": 0.12988416850566864, "learning_rate": 1e-06, "loss": 0.0009, "num_tokens": 58129647.0, "reward": 0.2421875, "reward_std": 0.15551914274692535, "rewards/simpleverify_reward/mean": 0.2421875, "rewards/simpleverify_reward/std": 0.4292463958263397, "step": 121, "tools/generated_tokens": 5552.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.02734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1336.1171875, "completions/mean_terminated_length": 1018.3841552734375, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.3691992927342653, "epoch": 0.020789400813683516, "frac_reward_zero_std": 0.375, "grad_norm": 0.17354363203048706, "learning_rate": 1e-06, "loss": 0.0241, "num_tokens": 58562637.0, "reward": 0.453125, "reward_std": 0.2553790807723999, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 122, "tools/generated_tokens": 5216.140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.89453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1317.43359375, "completions/mean_terminated_length": 997.2977905273438, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.34784369356930256, "epoch": 0.02095980573838584, "frac_reward_zero_std": 0.1875, "grad_norm": 0.19032245874404907, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 58985628.0, "reward": 0.296875, "reward_std": 0.29221853613853455, "rewards/simpleverify_reward/mean": 0.296875, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 123, "tools/generated_tokens": 5245.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.31640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1398.69921875, "completions/mean_terminated_length": 1098.1656494140625, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.3271794207394123, "epoch": 0.021130210663088164, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1481109857559204, "learning_rate": 1e-06, "loss": 0.0452, "num_tokens": 59430687.0, "reward": 0.4375, "reward_std": 0.26345524191856384, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 124, "tools/generated_tokens": 5726.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1288.40234375, "completions/mean_terminated_length": 943.14208984375, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.3033269513398409, "epoch": 0.02130061558779049, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14241783320903778, "learning_rate": 1e-06, "loss": 0.0136, "num_tokens": 59849270.0, "reward": 0.43359375, "reward_std": 0.21863040328025818, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 125, "tools/generated_tokens": 5112.4140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1443.78125, "completions/mean_terminated_length": 1104.8353271484375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.3240698855370283, "epoch": 0.021471020512492812, "frac_reward_zero_std": 0.375, "grad_norm": 0.13430829346179962, "learning_rate": 1e-06, "loss": -0.017, "num_tokens": 60294878.0, "reward": 0.37109375, "reward_std": 0.23877215385437012, "rewards/simpleverify_reward/mean": 0.37109375, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 126, "tools/generated_tokens": 5019.796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.74609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.31640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1436.2734375, "completions/mean_terminated_length": 1153.142822265625, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.3526413217186928, "epoch": 0.021641425437195137, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13777115941047668, "learning_rate": 1e-06, "loss": 0.0343, "num_tokens": 60749540.0, "reward": 0.31640625, "reward_std": 0.24117998778820038, "rewards/simpleverify_reward/mean": 0.31640625, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 127, "tools/generated_tokens": 5444.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.33203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1405.9296875, "completions/mean_terminated_length": 1086.77783203125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.32197364047169685, "epoch": 0.021811830361897457, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16866663098335266, "learning_rate": 1e-06, "loss": 0.0199, "num_tokens": 61200162.0, "reward": 0.44140625, "reward_std": 0.32650285959243774, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 128, "tools/generated_tokens": 5333.94140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1592.5, "completions/mean_terminated_length": 1099.9755859375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.33435916900634766, "epoch": 0.02198223528659978, "frac_reward_zero_std": 0.4375, "grad_norm": 0.12271421402692795, "learning_rate": 1e-06, "loss": 0.064, "num_tokens": 61697746.0, "reward": 0.234375, "reward_std": 0.19970625638961792, "rewards/simpleverify_reward/mean": 0.234375, "rewards/simpleverify_reward/std": 0.42443734407424927, "step": 129, "tools/generated_tokens": 6416.50390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1537.34375, "completions/mean_terminated_length": 1269.857177734375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.30458197370171547, "epoch": 0.022152640211302106, "frac_reward_zero_std": 0.375, "grad_norm": 0.1821925938129425, "learning_rate": 1e-06, "loss": 0.0457, "num_tokens": 62171482.0, "reward": 0.484375, "reward_std": 0.26566585898399353, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 130, "tools/generated_tokens": 5737.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.05078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1353.23828125, "completions/mean_terminated_length": 1116.801025390625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.28691938519477844, "epoch": 0.02232304513600443, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13503485918045044, "learning_rate": 1e-06, "loss": -0.0048, "num_tokens": 62599591.0, "reward": 0.49609375, "reward_std": 0.21896778047084808, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 131, "tools/generated_tokens": 4673.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1494.06640625, "completions/mean_terminated_length": 1246.83056640625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2928379736840725, "epoch": 0.022493450060706754, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11686106026172638, "learning_rate": 1e-06, "loss": -0.0004, "num_tokens": 63053000.0, "reward": 0.37890625, "reward_std": 0.1468954086303711, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 132, "tools/generated_tokens": 4662.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.37109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1466.23046875, "completions/mean_terminated_length": 1122.9503173828125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.32553007639944553, "epoch": 0.022663854985409078, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13065077364444733, "learning_rate": 1e-06, "loss": 0.0307, "num_tokens": 63515715.0, "reward": 0.26953125, "reward_std": 0.22028234601020813, "rewards/simpleverify_reward/mean": 0.26953125, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 133, "tools/generated_tokens": 5786.23828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1266.171875, "completions/mean_terminated_length": 1099.4312744140625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.3138121534138918, "epoch": 0.022834259910111402, "frac_reward_zero_std": 0.375, "grad_norm": 0.24351127445697784, "learning_rate": 1e-06, "loss": -0.0215, "num_tokens": 63922639.0, "reward": 0.52734375, "reward_std": 0.20970112085342407, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 134, "tools/generated_tokens": 4794.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.39453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1531.703125, "completions/mean_terminated_length": 1195.322509765625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.32346850633621216, "epoch": 0.023004664834813726, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11104744672775269, "learning_rate": 1e-06, "loss": 0.037, "num_tokens": 64404467.0, "reward": 0.26171875, "reward_std": 0.20310088992118835, "rewards/simpleverify_reward/mean": 0.26171875, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 135, "tools/generated_tokens": 5963.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.41015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1487.28515625, "completions/mean_terminated_length": 1097.3973388671875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.37246280163526535, "epoch": 0.02317506975951605, "frac_reward_zero_std": 0.5, "grad_norm": 0.12094295769929886, "learning_rate": 1e-06, "loss": 0.0145, "num_tokens": 64871292.0, "reward": 0.390625, "reward_std": 0.16691282391548157, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 136, "tools/generated_tokens": 5767.296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.08984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1332.03125, "completions/mean_terminated_length": 1088.3822021484375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.3254028670489788, "epoch": 0.023345474684218374, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1289188116788864, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 65292916.0, "reward": 0.3984375, "reward_std": 0.17835843563079834, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 137, "tools/generated_tokens": 4796.04296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1414.75390625, "completions/mean_terminated_length": 1077.2755126953125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.2777953064069152, "epoch": 0.0235158796089207, "frac_reward_zero_std": 0.5, "grad_norm": 0.12421082705259323, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 65737413.0, "reward": 0.46484375, "reward_std": 0.2040461003780365, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 138, "tools/generated_tokens": 5150.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1241.41796875, "completions/mean_terminated_length": 972.5573120117188, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.314918152987957, "epoch": 0.023686284533623023, "frac_reward_zero_std": 0.25, "grad_norm": 0.20149211585521698, "learning_rate": 1e-06, "loss": 0.0769, "num_tokens": 66141392.0, "reward": 0.54296875, "reward_std": 0.3109705150127411, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 139, "tools/generated_tokens": 4761.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1349.16015625, "completions/mean_terminated_length": 1048.5418701171875, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.3121089041233063, "epoch": 0.023856689458325347, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1447082757949829, "learning_rate": 1e-06, "loss": 0.0228, "num_tokens": 66567529.0, "reward": 0.38671875, "reward_std": 0.2085040807723999, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 140, "tools/generated_tokens": 5141.171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.41796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1472.1796875, "completions/mean_terminated_length": 1058.671142578125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.3121817819774151, "epoch": 0.02402709438302767, "frac_reward_zero_std": 0.375, "grad_norm": 0.17078803479671478, "learning_rate": 1e-06, "loss": 0.0383, "num_tokens": 67032487.0, "reward": 0.3984375, "reward_std": 0.24063238501548767, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 141, "tools/generated_tokens": 5712.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1251.36328125, "completions/mean_terminated_length": 1076.8619384765625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.3102445937693119, "epoch": 0.024197499307729995, "frac_reward_zero_std": 0.125, "grad_norm": 0.21668432652950287, "learning_rate": 1e-06, "loss": 0.0228, "num_tokens": 67439636.0, "reward": 0.57421875, "reward_std": 0.34586799144744873, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 142, "tools/generated_tokens": 4779.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1396.328125, "completions/mean_terminated_length": 1049.030029296875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.3218431733548641, "epoch": 0.024367904232432316, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14836956560611725, "learning_rate": 1e-06, "loss": 0.0293, "num_tokens": 67886424.0, "reward": 0.33984375, "reward_std": 0.2125907838344574, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 143, "tools/generated_tokens": 5436.33203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1389.7734375, "completions/mean_terminated_length": 1122.1484375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.3523574620485306, "epoch": 0.02453830915713464, "frac_reward_zero_std": 0.1875, "grad_norm": 0.20129433274269104, "learning_rate": 1e-06, "loss": 0.0397, "num_tokens": 68333518.0, "reward": 0.42578125, "reward_std": 0.3636796474456787, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 144, "tools/generated_tokens": 5765.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1398.36328125, "completions/mean_terminated_length": 1153.8763427734375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.2895941939204931, "epoch": 0.024708714081836964, "frac_reward_zero_std": 0.25, "grad_norm": 0.18979892134666443, "learning_rate": 1e-06, "loss": -0.0006, "num_tokens": 68777403.0, "reward": 0.4921875, "reward_std": 0.30592674016952515, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 145, "tools/generated_tokens": 5462.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1372.703125, "completions/mean_terminated_length": 993.8840942382812, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.3500816449522972, "epoch": 0.024879119006539288, "frac_reward_zero_std": 0.375, "grad_norm": 0.1640709489583969, "learning_rate": 1e-06, "loss": -0.0275, "num_tokens": 69222111.0, "reward": 0.23828125, "reward_std": 0.23751798272132874, "rewards/simpleverify_reward/mean": 0.23828125, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 146, "tools/generated_tokens": 5380.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1312.94140625, "completions/mean_terminated_length": 1078.0257568359375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.3114693034440279, "epoch": 0.025049523931241612, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2001684457063675, "learning_rate": 1e-06, "loss": -0.0069, "num_tokens": 69645680.0, "reward": 0.5, "reward_std": 0.2634032666683197, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 147, "tools/generated_tokens": 5088.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1553.296875, "completions/mean_terminated_length": 1266.25927734375, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.33421515114605427, "epoch": 0.025219928855943936, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1346120834350586, "learning_rate": 1e-06, "loss": 0.0333, "num_tokens": 70129516.0, "reward": 0.33203125, "reward_std": 0.25648343563079834, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 148, "tools/generated_tokens": 5561.3125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1350.3125, "completions/mean_terminated_length": 1021.5230102539062, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.2868925202637911, "epoch": 0.02539033378064626, "frac_reward_zero_std": 0.625, "grad_norm": 0.14119477570056915, "learning_rate": 1e-06, "loss": 0.037, "num_tokens": 70550860.0, "reward": 0.41015625, "reward_std": 0.1461106687784195, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 149, "tools/generated_tokens": 4838.33203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1988.0, "completions/mean_length": 1469.46875, "completions/mean_terminated_length": 1073.63818359375, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.3634101618081331, "epoch": 0.025560738705348585, "frac_reward_zero_std": 0.375, "grad_norm": 0.1538143754005432, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 71014340.0, "reward": 0.35546875, "reward_std": 0.2559266984462738, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 150, "tools/generated_tokens": 5749.46875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.08984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1345.05078125, "completions/mean_terminated_length": 1064.6392822265625, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.3609559182077646, "epoch": 0.02573114363005091, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17284370958805084, "learning_rate": 1e-06, "loss": 0.0424, "num_tokens": 71442577.0, "reward": 0.44140625, "reward_std": 0.22106516361236572, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 151, "tools/generated_tokens": 5217.0546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1619.13671875, "completions/mean_terminated_length": 1228.6865234375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.3010506443679333, "epoch": 0.025901548554753233, "frac_reward_zero_std": 0.375, "grad_norm": 0.11869866400957108, "learning_rate": 1e-06, "loss": 0.0247, "num_tokens": 71944500.0, "reward": 0.33203125, "reward_std": 0.24579845368862152, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 152, "tools/generated_tokens": 6203.140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1451.17578125, "completions/mean_terminated_length": 1138.5595703125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.3158528581261635, "epoch": 0.026071953479455557, "frac_reward_zero_std": 0.75, "grad_norm": 0.09035536646842957, "learning_rate": 1e-06, "loss": 0.0407, "num_tokens": 72399953.0, "reward": 0.38671875, "reward_std": 0.10409127175807953, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 153, "tools/generated_tokens": 5299.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1399.078125, "completions/mean_terminated_length": 1159.6417236328125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.37316756322979927, "epoch": 0.02624235840415788, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16092449426651, "learning_rate": 1e-06, "loss": 0.0123, "num_tokens": 72848549.0, "reward": 0.40234375, "reward_std": 0.2811351418495178, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 154, "tools/generated_tokens": 5431.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.96875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.51953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1629.77734375, "completions/mean_terminated_length": 1177.5528564453125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.3274534326046705, "epoch": 0.026412763328860205, "frac_reward_zero_std": 0.5625, "grad_norm": 0.09790311753749847, "learning_rate": 1e-06, "loss": 0.0033, "num_tokens": 73351500.0, "reward": 0.23828125, "reward_std": 0.18463993072509766, "rewards/simpleverify_reward/mean": 0.23828125, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 155, "tools/generated_tokens": 6509.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1213.96875, "completions/mean_terminated_length": 1059.5185546875, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.28100949712097645, "epoch": 0.02658316825356253, "frac_reward_zero_std": 0.25, "grad_norm": 0.15801389515399933, "learning_rate": 1e-06, "loss": 0.0149, "num_tokens": 73750612.0, "reward": 0.453125, "reward_std": 0.2885051667690277, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 156, "tools/generated_tokens": 4717.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 1998.0, "completions/mean_length": 1363.63671875, "completions/mean_terminated_length": 1154.142822265625, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.30511037074029446, "epoch": 0.026753573178264854, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14587201178073883, "learning_rate": 1e-06, "loss": -0.0028, "num_tokens": 74186647.0, "reward": 0.44140625, "reward_std": 0.18435022234916687, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 157, "tools/generated_tokens": 5235.65234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1257.265625, "completions/mean_terminated_length": 1102.079345703125, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.2574101975187659, "epoch": 0.026923978102967174, "frac_reward_zero_std": 0.5, "grad_norm": 0.16285882890224457, "learning_rate": 1e-06, "loss": -0.011, "num_tokens": 74583339.0, "reward": 0.65625, "reward_std": 0.18364217877388, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 158, "tools/generated_tokens": 3905.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1388.25390625, "completions/mean_terminated_length": 1114.884033203125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.26684923097491264, "epoch": 0.0270943830276695, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1531786024570465, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 75012652.0, "reward": 0.40625, "reward_std": 0.2531684637069702, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 159, "tools/generated_tokens": 4580.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1539.57421875, "completions/mean_terminated_length": 1131.4013671875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.3059833236038685, "epoch": 0.027264787952371822, "frac_reward_zero_std": 0.5, "grad_norm": 0.12701913714408875, "learning_rate": 1e-06, "loss": 0.0039, "num_tokens": 75494751.0, "reward": 0.35546875, "reward_std": 0.16625863313674927, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 160, "tools/generated_tokens": 5995.578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1390.21484375, "completions/mean_terminated_length": 1180.0, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.3158688638359308, "epoch": 0.027435192877074147, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16224777698516846, "learning_rate": 1e-06, "loss": 0.0048, "num_tokens": 75931430.0, "reward": 0.3125, "reward_std": 0.3364320993423462, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 161, "tools/generated_tokens": 5494.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.00390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1517.30859375, "completions/mean_terminated_length": 1018.7954711914062, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.29948427714407444, "epoch": 0.02760559780177647, "frac_reward_zero_std": 0.5625, "grad_norm": 0.16643038392066956, "learning_rate": 1e-06, "loss": 0.026, "num_tokens": 76405829.0, "reward": 0.33203125, "reward_std": 0.19893452525138855, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 162, "tools/generated_tokens": 6005.31640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1397.671875, "completions/mean_terminated_length": 1143.2010498046875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.3086371049284935, "epoch": 0.027776002726478795, "frac_reward_zero_std": 0.375, "grad_norm": 0.14533433318138123, "learning_rate": 1e-06, "loss": 0.0239, "num_tokens": 76848305.0, "reward": 0.33203125, "reward_std": 0.2884256839752197, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 163, "tools/generated_tokens": 5277.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.89453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1483.92578125, "completions/mean_terminated_length": 1198.5823974609375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.2766151363030076, "epoch": 0.02794640765118112, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17099782824516296, "learning_rate": 1e-06, "loss": 0.017, "num_tokens": 77313134.0, "reward": 0.41796875, "reward_std": 0.2649644613265991, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 164, "tools/generated_tokens": 5291.9453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.44140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1485.609375, "completions/mean_terminated_length": 1041.2098388671875, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.3144151847809553, "epoch": 0.028116812575883443, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1765998750925064, "learning_rate": 1e-06, "loss": 0.0347, "num_tokens": 77787674.0, "reward": 0.27734375, "reward_std": 0.24800434708595276, "rewards/simpleverify_reward/mean": 0.27734375, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 165, "tools/generated_tokens": 6005.6171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1479.140625, "completions/mean_terminated_length": 1211.063232421875, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.31386564671993256, "epoch": 0.028287217500585767, "frac_reward_zero_std": 0.25, "grad_norm": 0.15623007714748383, "learning_rate": 1e-06, "loss": 0.0348, "num_tokens": 78249838.0, "reward": 0.42578125, "reward_std": 0.2801070213317871, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 166, "tools/generated_tokens": 5519.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1282.81640625, "completions/mean_terminated_length": 1022.4136352539062, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.36202933825552464, "epoch": 0.02845762242528809, "frac_reward_zero_std": 0.6875, "grad_norm": 0.11055434495210648, "learning_rate": 1e-06, "loss": 0.0032, "num_tokens": 78651695.0, "reward": 0.328125, "reward_std": 0.1186390072107315, "rewards/simpleverify_reward/mean": 0.328125, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 167, "tools/generated_tokens": 4554.82421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2009.0, "completions/mean_length": 1410.91015625, "completions/mean_terminated_length": 1175.834228515625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.3066523037850857, "epoch": 0.028628027349990415, "frac_reward_zero_std": 0.25, "grad_norm": 0.14977262914180756, "learning_rate": 1e-06, "loss": 0.0274, "num_tokens": 79099000.0, "reward": 0.51171875, "reward_std": 0.30647432804107666, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 168, "tools/generated_tokens": 4818.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.37890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1478.41796875, "completions/mean_terminated_length": 1130.943359375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.2944907881319523, "epoch": 0.02879843227469274, "frac_reward_zero_std": 0.375, "grad_norm": 0.14796318113803864, "learning_rate": 1e-06, "loss": 0.0249, "num_tokens": 79562003.0, "reward": 0.36328125, "reward_std": 0.22028234601020813, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 169, "tools/generated_tokens": 5750.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.27734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1346.90234375, "completions/mean_terminated_length": 1077.83251953125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.33489724062383175, "epoch": 0.028968837199395064, "frac_reward_zero_std": 0.125, "grad_norm": 0.30806589126586914, "learning_rate": 1e-06, "loss": 0.0322, "num_tokens": 80003418.0, "reward": 0.3359375, "reward_std": 0.34549540281295776, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 170, "tools/generated_tokens": 5234.921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1499.08984375, "completions/mean_terminated_length": 1211.5714111328125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3544781617820263, "epoch": 0.029139242124097388, "frac_reward_zero_std": 0.25, "grad_norm": 0.16204816102981567, "learning_rate": 1e-06, "loss": 0.0064, "num_tokens": 80472081.0, "reward": 0.3984375, "reward_std": 0.31676173210144043, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 171, "tools/generated_tokens": 5459.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.36328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1476.9296875, "completions/mean_terminated_length": 1151.104248046875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.34173163399100304, "epoch": 0.029309647048799712, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1564481258392334, "learning_rate": 1e-06, "loss": 0.0499, "num_tokens": 80943743.0, "reward": 0.27734375, "reward_std": 0.31761646270751953, "rewards/simpleverify_reward/mean": 0.27734375, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 172, "tools/generated_tokens": 6148.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1213.7734375, "completions/mean_terminated_length": 1040.632080078125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.3498959634453058, "epoch": 0.029480051973502033, "frac_reward_zero_std": 0.25, "grad_norm": 0.1797487586736679, "learning_rate": 1e-06, "loss": -0.011, "num_tokens": 81338917.0, "reward": 0.3984375, "reward_std": 0.2843528985977173, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 173, "tools/generated_tokens": 4613.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1352.1953125, "completions/mean_terminated_length": 1052.8826904296875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.3132346123456955, "epoch": 0.029650456898204357, "frac_reward_zero_std": 0.375, "grad_norm": 0.16205665469169617, "learning_rate": 1e-06, "loss": 0.0375, "num_tokens": 81769623.0, "reward": 0.40234375, "reward_std": 0.26075831055641174, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 174, "tools/generated_tokens": 5056.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1256.58984375, "completions/mean_terminated_length": 1127.0863037109375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.28759870771318674, "epoch": 0.02982086182290668, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14383946359157562, "learning_rate": 1e-06, "loss": 0.0142, "num_tokens": 82181790.0, "reward": 0.44921875, "reward_std": 0.1560128629207611, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 175, "tools/generated_tokens": 4712.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1320.77734375, "completions/mean_terminated_length": 1126.3812255859375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.29212189465761185, "epoch": 0.029991266747609005, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13572239875793457, "learning_rate": 1e-06, "loss": -0.0069, "num_tokens": 82609605.0, "reward": 0.49609375, "reward_std": 0.19864007830619812, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 176, "tools/generated_tokens": 4904.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1314.265625, "completions/mean_terminated_length": 1136.1748046875, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.3161185160279274, "epoch": 0.03016167167231133, "frac_reward_zero_std": 0.125, "grad_norm": 0.2562132477760315, "learning_rate": 1e-06, "loss": -0.0042, "num_tokens": 83031017.0, "reward": 0.5078125, "reward_std": 0.3289920687675476, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 177, "tools/generated_tokens": 4546.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1411.23828125, "completions/mean_terminated_length": 1121.8125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.34062889590859413, "epoch": 0.030332076597013653, "frac_reward_zero_std": 0.6875, "grad_norm": 0.10277829319238663, "learning_rate": 1e-06, "loss": 0.0138, "num_tokens": 83473494.0, "reward": 0.33984375, "reward_std": 0.10596734285354614, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 178, "tools/generated_tokens": 4875.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1350.0703125, "completions/mean_terminated_length": 1150.1708984375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.32410680316388607, "epoch": 0.030502481521715977, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15116848051548004, "learning_rate": 1e-06, "loss": 0.0147, "num_tokens": 83901896.0, "reward": 0.41796875, "reward_std": 0.21777918934822083, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 179, "tools/generated_tokens": 4670.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1518.0859375, "completions/mean_terminated_length": 1167.1038818359375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.36199636943638325, "epoch": 0.0306728864464183, "frac_reward_zero_std": 0.5, "grad_norm": 0.13650768995285034, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 84370446.0, "reward": 0.31640625, "reward_std": 0.1536140739917755, "rewards/simpleverify_reward/mean": 0.31640625, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 180, "tools/generated_tokens": 5438.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1356.03515625, "completions/mean_terminated_length": 1162.2850341796875, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.341150039806962, "epoch": 0.030843291371120626, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1136154979467392, "learning_rate": 1e-06, "loss": 0.0224, "num_tokens": 84797719.0, "reward": 0.5234375, "reward_std": 0.15325656533241272, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 181, "tools/generated_tokens": 4316.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1319.46484375, "completions/mean_terminated_length": 1219.0888671875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.2949541173875332, "epoch": 0.03101369629582295, "frac_reward_zero_std": 0.375, "grad_norm": 0.15052412450313568, "learning_rate": 1e-06, "loss": 0.0026, "num_tokens": 85216174.0, "reward": 0.6171875, "reward_std": 0.2403016835451126, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 182, "tools/generated_tokens": 4583.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1295.13671875, "completions/mean_terminated_length": 1116.9227294921875, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.3051509000360966, "epoch": 0.031184101220525274, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1615077406167984, "learning_rate": 1e-06, "loss": 0.0061, "num_tokens": 85632577.0, "reward": 0.51953125, "reward_std": 0.27994656562805176, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 183, "tools/generated_tokens": 4967.140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.79296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2010.0, "completions/mean_length": 1411.5234375, "completions/mean_terminated_length": 1229.2210693359375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.32412454672157764, "epoch": 0.0313545061452276, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1103578731417656, "learning_rate": 1e-06, "loss": 0.0123, "num_tokens": 86067063.0, "reward": 0.3671875, "reward_std": 0.16133463382720947, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 184, "tools/generated_tokens": 4707.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1334.00390625, "completions/mean_terminated_length": 1075.7552490234375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.32885063998401165, "epoch": 0.03152491106992992, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17355377972126007, "learning_rate": 1e-06, "loss": -0.0016, "num_tokens": 86491688.0, "reward": 0.4453125, "reward_std": 0.2305552214384079, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 185, "tools/generated_tokens": 4966.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1390.4296875, "completions/mean_terminated_length": 1162.0106201171875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2966838479042053, "epoch": 0.031695315994632246, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1571619063615799, "learning_rate": 1e-06, "loss": -0.011, "num_tokens": 86927190.0, "reward": 0.5078125, "reward_std": 0.23590736091136932, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 186, "tools/generated_tokens": 4670.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1164.3203125, "completions/mean_terminated_length": 1093.4766845703125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.3238255549222231, "epoch": 0.03186572091933457, "frac_reward_zero_std": 0.375, "grad_norm": 0.17975440621376038, "learning_rate": 1e-06, "loss": -0.0092, "num_tokens": 87310472.0, "reward": 0.76171875, "reward_std": 0.2568175494670868, "rewards/simpleverify_reward/mean": 0.76171875, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 187, "tools/generated_tokens": 4012.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1378.62109375, "completions/mean_terminated_length": 1182.5404052734375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.3143183123320341, "epoch": 0.032036125844036895, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1601002961397171, "learning_rate": 1e-06, "loss": 0.0318, "num_tokens": 87745127.0, "reward": 0.453125, "reward_std": 0.2835540473461151, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 188, "tools/generated_tokens": 4778.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1188.34765625, "completions/mean_terminated_length": 1029.15283203125, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.3333643972873688, "epoch": 0.032206530768739215, "frac_reward_zero_std": 0.1875, "grad_norm": 0.168931782245636, "learning_rate": 1e-06, "loss": 0.016, "num_tokens": 88131792.0, "reward": 0.48828125, "reward_std": 0.302188515663147, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 189, "tools/generated_tokens": 4556.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.64453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.31640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1523.765625, "completions/mean_terminated_length": 1281.125732421875, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.3226275350898504, "epoch": 0.03237693569344154, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12362457811832428, "learning_rate": 1e-06, "loss": 0.0096, "num_tokens": 88596084.0, "reward": 0.3515625, "reward_std": 0.13896197080612183, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 190, "tools/generated_tokens": 5195.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.79296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 1998.0, "completions/mean_length": 1396.05078125, "completions/mean_terminated_length": 1125.91162109375, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3478570803999901, "epoch": 0.032547340618143863, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18209584057331085, "learning_rate": 1e-06, "loss": 0.044, "num_tokens": 89031665.0, "reward": 0.4453125, "reward_std": 0.2613418400287628, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 191, "tools/generated_tokens": 5108.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1321.42578125, "completions/mean_terminated_length": 1122.6119384765625, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.29072050005197525, "epoch": 0.03271774554284619, "frac_reward_zero_std": 0.25, "grad_norm": 0.16858156025409698, "learning_rate": 1e-06, "loss": 0.0301, "num_tokens": 89470286.0, "reward": 0.5234375, "reward_std": 0.30222654342651367, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 192, "tools/generated_tokens": 5169.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1489.44921875, "completions/mean_terminated_length": 1176.1219482421875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.29124719835817814, "epoch": 0.03288815046754851, "frac_reward_zero_std": 0.1875, "grad_norm": 0.27041056752204895, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 89932769.0, "reward": 0.47265625, "reward_std": 0.3046509623527527, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 193, "tools/generated_tokens": 5553.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1351.3671875, "completions/mean_terminated_length": 1123.9688720703125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.3492069635540247, "epoch": 0.03305855539225084, "frac_reward_zero_std": 0.25, "grad_norm": 0.18806102871894836, "learning_rate": 1e-06, "loss": 0.0223, "num_tokens": 90372047.0, "reward": 0.5546875, "reward_std": 0.2830354869365692, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 194, "tools/generated_tokens": 5111.3671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1253.56640625, "completions/mean_terminated_length": 1167.5887451171875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2873858269304037, "epoch": 0.03322896031695316, "frac_reward_zero_std": 0.3125, "grad_norm": 0.14380556344985962, "learning_rate": 1e-06, "loss": 0.0323, "num_tokens": 90775760.0, "reward": 0.49609375, "reward_std": 0.26275384426116943, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 195, "tools/generated_tokens": 4005.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1291.87890625, "completions/mean_terminated_length": 1139.2347412109375, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.3549950644373894, "epoch": 0.03339936524165548, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1846829056739807, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 91187697.0, "reward": 0.40234375, "reward_std": 0.33350884914398193, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 196, "tools/generated_tokens": 4331.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1270.66015625, "completions/mean_terminated_length": 1139.3287353515625, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.31027913466095924, "epoch": 0.03356977016635781, "frac_reward_zero_std": 0.625, "grad_norm": 0.1424965262413025, "learning_rate": 1e-06, "loss": -0.0202, "num_tokens": 91596890.0, "reward": 0.5, "reward_std": 0.1507449597120285, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 197, "tools/generated_tokens": 4342.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.41796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2000.0, "completions/mean_length": 1501.75, "completions/mean_terminated_length": 1109.4765625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3565778099000454, "epoch": 0.03374017509106013, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16838455200195312, "learning_rate": 1e-06, "loss": 0.0645, "num_tokens": 92066794.0, "reward": 0.359375, "reward_std": 0.2964656949043274, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 198, "tools/generated_tokens": 5725.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1269.703125, "completions/mean_terminated_length": 1121.28369140625, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.3502108883112669, "epoch": 0.033910580015762457, "frac_reward_zero_std": 0.1875, "grad_norm": 0.20792637765407562, "learning_rate": 1e-06, "loss": 0.0462, "num_tokens": 92480238.0, "reward": 0.5625, "reward_std": 0.3204492926597595, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 199, "tools/generated_tokens": 5061.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1424.1875, "completions/mean_terminated_length": 1085.98193359375, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.3356624115258455, "epoch": 0.03408098494046478, "frac_reward_zero_std": 0.375, "grad_norm": 0.24913813173770905, "learning_rate": 1e-06, "loss": 0.0232, "num_tokens": 92933614.0, "reward": 0.2890625, "reward_std": 0.2746789753437042, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 200, "tools/generated_tokens": 5440.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1463.45703125, "completions/mean_terminated_length": 1050.38671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.3088175095617771, "epoch": 0.034251389865167105, "frac_reward_zero_std": 0.25, "grad_norm": 0.15846951305866241, "learning_rate": 1e-06, "loss": 0.0449, "num_tokens": 93392115.0, "reward": 0.2890625, "reward_std": 0.30979403853416443, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 201, "tools/generated_tokens": 5623.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.03125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1363.51953125, "completions/mean_terminated_length": 1217.540283203125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2520632538944483, "epoch": 0.034421794789869425, "frac_reward_zero_std": 0.375, "grad_norm": 0.12616130709648132, "learning_rate": 1e-06, "loss": -0.0291, "num_tokens": 93819400.0, "reward": 0.61328125, "reward_std": 0.21367931365966797, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 202, "tools/generated_tokens": 3867.53125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1993.0, "completions/mean_length": 1499.68359375, "completions/mean_terminated_length": 1159.594970703125, "completions/min_length": 318.0, "completions/min_terminated_length": 318.0, "entropy": 0.34026093780994415, "epoch": 0.03459219971457175, "frac_reward_zero_std": 0.5, "grad_norm": 0.1201467216014862, "learning_rate": 1e-06, "loss": 0.0179, "num_tokens": 94287399.0, "reward": 0.203125, "reward_std": 0.17978152632713318, "rewards/simpleverify_reward/mean": 0.203125, "rewards/simpleverify_reward/std": 0.40311288833618164, "step": 203, "tools/generated_tokens": 5587.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1325.0, "completions/mean_terminated_length": 1042.0870361328125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.3588677067309618, "epoch": 0.034762604639274074, "frac_reward_zero_std": 0.375, "grad_norm": 0.1788933426141739, "learning_rate": 1e-06, "loss": 0.0247, "num_tokens": 94714951.0, "reward": 0.32421875, "reward_std": 0.2567910850048065, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 204, "tools/generated_tokens": 4909.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1116.29296875, "completions/mean_terminated_length": 1033.0340576171875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.31627057492733, "epoch": 0.0349330095639764, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16822679340839386, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 95078178.0, "reward": 0.6015625, "reward_std": 0.28749823570251465, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 205, "tools/generated_tokens": 3820.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1457.61328125, "completions/mean_terminated_length": 1148.3690185546875, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.29588034749031067, "epoch": 0.03510341448867872, "frac_reward_zero_std": 0.375, "grad_norm": 0.14018484950065613, "learning_rate": 1e-06, "loss": -0.0083, "num_tokens": 95551183.0, "reward": 0.3828125, "reward_std": 0.2541801333427429, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 206, "tools/generated_tokens": 5465.62890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1238.25, "completions/mean_terminated_length": 1092.7188720703125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.3713560700416565, "epoch": 0.03527381941338105, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15887637436389923, "learning_rate": 1e-06, "loss": -0.0072, "num_tokens": 95948911.0, "reward": 0.53515625, "reward_std": 0.24932172894477844, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 207, "tools/generated_tokens": 4262.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1590.796875, "completions/mean_terminated_length": 1199.8551025390625, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.38159251026809216, "epoch": 0.03544422433808337, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16650021076202393, "learning_rate": 1e-06, "loss": 0.0254, "num_tokens": 96445083.0, "reward": 0.21875, "reward_std": 0.26362934708595276, "rewards/simpleverify_reward/mean": 0.21875, "rewards/simpleverify_reward/std": 0.41420844197273254, "step": 208, "tools/generated_tokens": 6430.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1264.2421875, "completions/mean_terminated_length": 1064.4608154296875, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.27137147448956966, "epoch": 0.0356146292627857, "frac_reward_zero_std": 0.375, "grad_norm": 0.1762692630290985, "learning_rate": 1e-06, "loss": 0.0067, "num_tokens": 96849385.0, "reward": 0.44921875, "reward_std": 0.23018452525138855, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 209, "tools/generated_tokens": 4240.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1158.6640625, "completions/mean_terminated_length": 1022.45947265625, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.2671739049255848, "epoch": 0.03578503418748802, "frac_reward_zero_std": 0.25, "grad_norm": 0.177192822098732, "learning_rate": 1e-06, "loss": 0.032, "num_tokens": 97225027.0, "reward": 0.359375, "reward_std": 0.30830952525138855, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 210, "tools/generated_tokens": 4030.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1336.9375, "completions/mean_terminated_length": 1146.8663330078125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.29814364202320576, "epoch": 0.03595543911219034, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1752936989068985, "learning_rate": 1e-06, "loss": 0.0314, "num_tokens": 97642931.0, "reward": 0.47265625, "reward_std": 0.2892768681049347, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 211, "tools/generated_tokens": 4336.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1254.73046875, "completions/mean_terminated_length": 1172.6680908203125, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.28304185532033443, "epoch": 0.03612584403689267, "frac_reward_zero_std": 0.125, "grad_norm": 0.18348625302314758, "learning_rate": 1e-06, "loss": 0.0053, "num_tokens": 98053182.0, "reward": 0.50390625, "reward_std": 0.35764625668525696, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 212, "tools/generated_tokens": 4198.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1418.8984375, "completions/mean_terminated_length": 1200.373779296875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.3266041036695242, "epoch": 0.03629624896159499, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17745855450630188, "learning_rate": 1e-06, "loss": 0.0266, "num_tokens": 98498468.0, "reward": 0.4921875, "reward_std": 0.2730247974395752, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 213, "tools/generated_tokens": 4882.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1392.9921875, "completions/mean_terminated_length": 1271.6990966796875, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.28982884902507067, "epoch": 0.036466653886297315, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15318572521209717, "learning_rate": 1e-06, "loss": 0.042, "num_tokens": 98929938.0, "reward": 0.58984375, "reward_std": 0.2049104869365692, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 214, "tools/generated_tokens": 4409.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1411.09375, "completions/mean_terminated_length": 1198.796875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.3525862656533718, "epoch": 0.036637058810999636, "frac_reward_zero_std": 0.5, "grad_norm": 0.15535373985767365, "learning_rate": 1e-06, "loss": -0.0059, "num_tokens": 99368538.0, "reward": 0.34375, "reward_std": 0.17493700981140137, "rewards/simpleverify_reward/mean": 0.34375, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 215, "tools/generated_tokens": 4867.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1136.0234375, "completions/mean_terminated_length": 1083.272705078125, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.2982969619333744, "epoch": 0.03680746373570196, "frac_reward_zero_std": 0.25, "grad_norm": 0.16756686568260193, "learning_rate": 1e-06, "loss": 0.0036, "num_tokens": 99744672.0, "reward": 0.453125, "reward_std": 0.316123366355896, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 216, "tools/generated_tokens": 3888.01953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1372.48828125, "completions/mean_terminated_length": 1179.010009765625, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.30694323405623436, "epoch": 0.036977868660404284, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16647158563137054, "learning_rate": 1e-06, "loss": 0.0192, "num_tokens": 100172189.0, "reward": 0.4921875, "reward_std": 0.2452508509159088, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 217, "tools/generated_tokens": 4300.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1289.03125, "completions/mean_terminated_length": 1066.70703125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.3073802124708891, "epoch": 0.03714827358510661, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13852950930595398, "learning_rate": 1e-06, "loss": 0.0238, "num_tokens": 100583205.0, "reward": 0.41796875, "reward_std": 0.169600710272789, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 218, "tools/generated_tokens": 4425.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1266.91015625, "completions/mean_terminated_length": 1086.6634521484375, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.2961250003427267, "epoch": 0.03731867850980893, "frac_reward_zero_std": 0.375, "grad_norm": 0.1482100486755371, "learning_rate": 1e-06, "loss": 0.0184, "num_tokens": 100992926.0, "reward": 0.5625, "reward_std": 0.26278093457221985, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 219, "tools/generated_tokens": 4330.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1334.78125, "completions/mean_terminated_length": 1157.3463134765625, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "entropy": 0.35707367956638336, "epoch": 0.03748908343451126, "frac_reward_zero_std": 0.25, "grad_norm": 0.17978939414024353, "learning_rate": 1e-06, "loss": 0.0039, "num_tokens": 101421190.0, "reward": 0.5703125, "reward_std": 0.31071737408638, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 220, "tools/generated_tokens": 5030.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.27734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1439.80859375, "completions/mean_terminated_length": 1206.3946533203125, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.3434401638805866, "epoch": 0.03765948835921358, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17241796851158142, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 101872101.0, "reward": 0.46875, "reward_std": 0.24075186252593994, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 221, "tools/generated_tokens": 5247.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1464.3984375, "completions/mean_terminated_length": 1304.7064208984375, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.3227778486907482, "epoch": 0.03782989328391591, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14050684869289398, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 102323163.0, "reward": 0.53515625, "reward_std": 0.24534353613853455, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 222, "tools/generated_tokens": 4824.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1236.2109375, "completions/mean_terminated_length": 971.2279663085938, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.30677394196391106, "epoch": 0.03800029820861823, "frac_reward_zero_std": 0.5, "grad_norm": 0.14632734656333923, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 102722961.0, "reward": 0.3984375, "reward_std": 0.21662378311157227, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 223, "tools/generated_tokens": 4924.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1288.19921875, "completions/mean_terminated_length": 1205.9697265625, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.2895997706800699, "epoch": 0.038170703133320556, "frac_reward_zero_std": 0.375, "grad_norm": 0.15224234759807587, "learning_rate": 1e-06, "loss": 0.0288, "num_tokens": 103124916.0, "reward": 0.5, "reward_std": 0.23778533935546875, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 224, "tools/generated_tokens": 3928.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1339.4375, "completions/mean_terminated_length": 1154.4482421875, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.29526018910109997, "epoch": 0.03834110805802288, "frac_reward_zero_std": 0.25, "grad_norm": 0.17124171555042267, "learning_rate": 1e-06, "loss": 0.023, "num_tokens": 103552052.0, "reward": 0.53515625, "reward_std": 0.27697813510894775, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 225, "tools/generated_tokens": 4547.453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1242.125, "completions/mean_terminated_length": 984.5824584960938, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.32621590234339237, "epoch": 0.0385115129827252, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19975169003009796, "learning_rate": 1e-06, "loss": 0.0191, "num_tokens": 103957508.0, "reward": 0.5, "reward_std": 0.25395601987838745, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 226, "tools/generated_tokens": 4698.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1308.50390625, "completions/mean_terminated_length": 990.4022216796875, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.3738710843026638, "epoch": 0.038681917907427525, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14983998239040375, "learning_rate": 1e-06, "loss": 0.016, "num_tokens": 104376181.0, "reward": 0.22265625, "reward_std": 0.1936889886856079, "rewards/simpleverify_reward/mean": 0.22265625, "rewards/simpleverify_reward/std": 0.41684433817863464, "step": 227, "tools/generated_tokens": 5020.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1273.3984375, "completions/mean_terminated_length": 1071.16748046875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.32002140395343304, "epoch": 0.038852322832129846, "frac_reward_zero_std": 0.375, "grad_norm": 0.14927567541599274, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 104783547.0, "reward": 0.515625, "reward_std": 0.2427476942539215, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 228, "tools/generated_tokens": 4609.40625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1447.65625, "completions/mean_terminated_length": 1133.1905517578125, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.3097304329276085, "epoch": 0.03902272775683217, "frac_reward_zero_std": 0.5, "grad_norm": 0.11527150869369507, "learning_rate": 1e-06, "loss": 0.0202, "num_tokens": 105240195.0, "reward": 0.390625, "reward_std": 0.17829003930091858, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 229, "tools/generated_tokens": 5359.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1183.796875, "completions/mean_terminated_length": 1110.559326171875, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.28834480978548527, "epoch": 0.039193132681534494, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16462694108486176, "learning_rate": 1e-06, "loss": 0.0295, "num_tokens": 105616607.0, "reward": 0.5234375, "reward_std": 0.266690731048584, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 230, "tools/generated_tokens": 3559.796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1300.5546875, "completions/mean_terminated_length": 1132.4688720703125, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.3693099822849035, "epoch": 0.03936353760623682, "frac_reward_zero_std": 0.1875, "grad_norm": 0.19829939305782318, "learning_rate": 1e-06, "loss": 0.0239, "num_tokens": 106034141.0, "reward": 0.3125, "reward_std": 0.30828261375427246, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 231, "tools/generated_tokens": 4892.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1380.33203125, "completions/mean_terminated_length": 1103.674072265625, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.36126304790377617, "epoch": 0.03953394253093914, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15957269072532654, "learning_rate": 1e-06, "loss": 0.0367, "num_tokens": 106475794.0, "reward": 0.30859375, "reward_std": 0.25956130027770996, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 232, "tools/generated_tokens": 5148.3359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.83984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.27734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1520.0546875, "completions/mean_terminated_length": 1317.4378662109375, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.27091052010655403, "epoch": 0.03970434745564147, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1620144098997116, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 106929008.0, "reward": 0.4296875, "reward_std": 0.2632311284542084, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 233, "tools/generated_tokens": 4392.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1309.3203125, "completions/mean_terminated_length": 1164.3597412109375, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.3271372374147177, "epoch": 0.03987475238034379, "frac_reward_zero_std": 0.375, "grad_norm": 0.17116455733776093, "learning_rate": 1e-06, "loss": 0.0132, "num_tokens": 107348258.0, "reward": 0.484375, "reward_std": 0.1986129879951477, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 234, "tools/generated_tokens": 3965.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1294.58203125, "completions/mean_terminated_length": 1083.625, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.32896302081644535, "epoch": 0.04004515730504612, "frac_reward_zero_std": 0.5, "grad_norm": 0.12428409606218338, "learning_rate": 1e-06, "loss": 0.0203, "num_tokens": 107764087.0, "reward": 0.34765625, "reward_std": 0.1701192855834961, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 235, "tools/generated_tokens": 4750.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1486.9140625, "completions/mean_terminated_length": 1193.011962890625, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.3677659723907709, "epoch": 0.04021556222974844, "frac_reward_zero_std": 0.5, "grad_norm": 0.13566716015338898, "learning_rate": 1e-06, "loss": 0.0076, "num_tokens": 108226577.0, "reward": 0.3359375, "reward_std": 0.19036275148391724, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 236, "tools/generated_tokens": 5510.91796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2000.0, "completions/mean_length": 1409.0390625, "completions/mean_terminated_length": 1217.685302734375, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.31223051622509956, "epoch": 0.040385967154450766, "frac_reward_zero_std": 0.25, "grad_norm": 0.16460387408733368, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 108679195.0, "reward": 0.31640625, "reward_std": 0.2826101779937744, "rewards/simpleverify_reward/mean": 0.31640625, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 237, "tools/generated_tokens": 5329.0546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1198.70703125, "completions/mean_terminated_length": 1085.968994140625, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.33975757844746113, "epoch": 0.04055637207915309, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13597051799297333, "learning_rate": 1e-06, "loss": 0.0111, "num_tokens": 109059792.0, "reward": 0.5390625, "reward_std": 0.1668444126844406, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 238, "tools/generated_tokens": 3646.7109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1248.859375, "completions/mean_terminated_length": 1118.0908203125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.3333216030150652, "epoch": 0.040726777003855415, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1780129075050354, "learning_rate": 1e-06, "loss": 0.0126, "num_tokens": 109461260.0, "reward": 0.625, "reward_std": 0.291700541973114, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 239, "tools/generated_tokens": 4096.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1264.7421875, "completions/mean_terminated_length": 1088.602783203125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.35745963640511036, "epoch": 0.040897181928557735, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17514276504516602, "learning_rate": 1e-06, "loss": 0.0093, "num_tokens": 109869178.0, "reward": 0.3828125, "reward_std": 0.29590702056884766, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 240, "tools/generated_tokens": 4968.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1181.03515625, "completions/mean_terminated_length": 1039.1680908203125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.3018411621451378, "epoch": 0.041067586853260056, "frac_reward_zero_std": 0.5, "grad_norm": 0.124259814620018, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 110277523.0, "reward": 0.4609375, "reward_std": 0.20026493072509766, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 241, "tools/generated_tokens": 4117.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1209.26953125, "completions/mean_terminated_length": 1067.566162109375, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.328345762565732, "epoch": 0.041237991777962384, "frac_reward_zero_std": 0.375, "grad_norm": 0.1596774160861969, "learning_rate": 1e-06, "loss": 0.027, "num_tokens": 110669880.0, "reward": 0.421875, "reward_std": 0.2532879114151001, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 242, "tools/generated_tokens": 4409.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1407.734375, "completions/mean_terminated_length": 1194.3177490234375, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.34485598281025887, "epoch": 0.041408396702664704, "frac_reward_zero_std": 0.5, "grad_norm": 0.13737072050571442, "learning_rate": 1e-06, "loss": -0.0124, "num_tokens": 111114804.0, "reward": 0.44921875, "reward_std": 0.19643138349056244, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 243, "tools/generated_tokens": 5007.74609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1274.12109375, "completions/mean_terminated_length": 1090.932373046875, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.2873286344110966, "epoch": 0.04157880162736703, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14881965517997742, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 111529747.0, "reward": 0.41796875, "reward_std": 0.2261517196893692, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 244, "tools/generated_tokens": 4586.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1275.21484375, "completions/mean_terminated_length": 1028.2421875, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.3055835347622633, "epoch": 0.04174920655206935, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13444218039512634, "learning_rate": 1e-06, "loss": 0.0466, "num_tokens": 111944778.0, "reward": 0.34765625, "reward_std": 0.23195403814315796, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 245, "tools/generated_tokens": 4803.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1265.47265625, "completions/mean_terminated_length": 1149.6727294921875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.3193060848861933, "epoch": 0.04191961147677168, "frac_reward_zero_std": 0.375, "grad_norm": 0.16008260846138, "learning_rate": 1e-06, "loss": 0.0306, "num_tokens": 112343955.0, "reward": 0.5, "reward_std": 0.2652543783187866, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 246, "tools/generated_tokens": 4273.484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1287.78515625, "completions/mean_terminated_length": 1112.3509521484375, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.28172095213085413, "epoch": 0.042090016401474, "frac_reward_zero_std": 0.375, "grad_norm": 0.16602839529514313, "learning_rate": 1e-06, "loss": 0.0289, "num_tokens": 112749276.0, "reward": 0.53515625, "reward_std": 0.20536066591739655, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 247, "tools/generated_tokens": 4207.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.42578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1444.0078125, "completions/mean_terminated_length": 1193.73486328125, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.35894401371479034, "epoch": 0.04226042132617633, "frac_reward_zero_std": 0.375, "grad_norm": 0.1456543207168579, "learning_rate": 1e-06, "loss": 0.0298, "num_tokens": 113198046.0, "reward": 0.4609375, "reward_std": 0.26840826869010925, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 248, "tools/generated_tokens": 5084.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1265.109375, "completions/mean_terminated_length": 1157.2445068359375, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.32613920606672764, "epoch": 0.04243082625087865, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1564120352268219, "learning_rate": 1e-06, "loss": 0.0121, "num_tokens": 113603210.0, "reward": 0.52734375, "reward_std": 0.25029462575912476, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 249, "tools/generated_tokens": 4433.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1351.50390625, "completions/mean_terminated_length": 1138.290771484375, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.30417640320956707, "epoch": 0.04260123117558098, "frac_reward_zero_std": 0.25, "grad_norm": 1.2026276588439941, "learning_rate": 1e-06, "loss": 0.0032, "num_tokens": 114037643.0, "reward": 0.43359375, "reward_std": 0.27749860286712646, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 250, "tools/generated_tokens": 4975.50390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1422.83203125, "completions/mean_terminated_length": 1158.8778076171875, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.3520346116274595, "epoch": 0.0427716361002833, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12789323925971985, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 114485888.0, "reward": 0.33984375, "reward_std": 0.1596985161304474, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 251, "tools/generated_tokens": 5254.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1381.640625, "completions/mean_terminated_length": 1120.896728515625, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.3559390101581812, "epoch": 0.042942041024985625, "frac_reward_zero_std": 0.25, "grad_norm": 0.164140984416008, "learning_rate": 1e-06, "loss": 0.0717, "num_tokens": 114925684.0, "reward": 0.2734375, "reward_std": 0.3424764573574066, "rewards/simpleverify_reward/mean": 0.2734375, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 252, "tools/generated_tokens": 5661.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.08984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1376.078125, "completions/mean_terminated_length": 1183.628173828125, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.2860397193580866, "epoch": 0.043112445949687946, "frac_reward_zero_std": 0.1875, "grad_norm": 0.15713639557361603, "learning_rate": 1e-06, "loss": 0.0416, "num_tokens": 115359624.0, "reward": 0.578125, "reward_std": 0.33410364389419556, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 253, "tools/generated_tokens": 4888.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1262.1171875, "completions/mean_terminated_length": 1149.852783203125, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.28291032928973436, "epoch": 0.04328285087439027, "frac_reward_zero_std": 0.3125, "grad_norm": 0.170853853225708, "learning_rate": 1e-06, "loss": -0.0058, "num_tokens": 115758870.0, "reward": 0.48046875, "reward_std": 0.28905272483825684, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 254, "tools/generated_tokens": 4078.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.31640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1479.47265625, "completions/mean_terminated_length": 1216.3314208984375, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.3204533886164427, "epoch": 0.043453255799092594, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17299295961856842, "learning_rate": 1e-06, "loss": 0.0274, "num_tokens": 116218159.0, "reward": 0.4296875, "reward_std": 0.33490437269210815, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 255, "tools/generated_tokens": 5439.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1321.59765625, "completions/mean_terminated_length": 1104.045654296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.29396906588226557, "epoch": 0.043623660723794914, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1581735461950302, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 116640632.0, "reward": 0.4453125, "reward_std": 0.2293090522289276, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 256, "tools/generated_tokens": 4609.6171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1195.21484375, "completions/mean_terminated_length": 1171.240966796875, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.25802111998200417, "epoch": 0.04379406564849724, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1504235863685608, "learning_rate": 1e-06, "loss": 0.0101, "num_tokens": 117019567.0, "reward": 0.77734375, "reward_std": 0.199052095413208, "rewards/simpleverify_reward/mean": 0.77734375, "rewards/simpleverify_reward/std": 0.41684433817863464, "step": 257, "tools/generated_tokens": 3523.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1347.33203125, "completions/mean_terminated_length": 1185.6395263671875, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.2890857020393014, "epoch": 0.04396447057319956, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14119593799114227, "learning_rate": 1e-06, "loss": 0.0315, "num_tokens": 117440452.0, "reward": 0.51171875, "reward_std": 0.1848640739917755, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 258, "tools/generated_tokens": 4323.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1279.1875, "completions/mean_terminated_length": 1177.1326904296875, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.29742043651640415, "epoch": 0.04413487549790189, "frac_reward_zero_std": 0.375, "grad_norm": 0.14363500475883484, "learning_rate": 1e-06, "loss": 0.0274, "num_tokens": 117851124.0, "reward": 0.51953125, "reward_std": 0.3115956783294678, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 259, "tools/generated_tokens": 4759.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1396.6953125, "completions/mean_terminated_length": 1151.5806884765625, "completions/min_length": 300.0, "completions/min_terminated_length": 300.0, "entropy": 0.33709784410893917, "epoch": 0.04430528042260421, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15909504890441895, "learning_rate": 1e-06, "loss": 0.0154, "num_tokens": 118284790.0, "reward": 0.44140625, "reward_std": 0.2815985083580017, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 260, "tools/generated_tokens": 4908.703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1325.0078125, "completions/mean_terminated_length": 1206.699951171875, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.26144256815314293, "epoch": 0.04447568534730654, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1711936742067337, "learning_rate": 1e-06, "loss": 0.002, "num_tokens": 118696616.0, "reward": 0.4296875, "reward_std": 0.27056455612182617, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 261, "tools/generated_tokens": 3893.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2012.0, "completions/mean_length": 1281.9609375, "completions/mean_terminated_length": 1127.3145751953125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.2766091823577881, "epoch": 0.04464609027200886, "frac_reward_zero_std": 0.375, "grad_norm": 0.13621380925178528, "learning_rate": 1e-06, "loss": -0.0175, "num_tokens": 119113022.0, "reward": 0.515625, "reward_std": 0.25350111722946167, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 262, "tools/generated_tokens": 4705.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1295.52734375, "completions/mean_terminated_length": 1191.8577880859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2822153940796852, "epoch": 0.04481649519671119, "frac_reward_zero_std": 0.25, "grad_norm": 0.17557425796985626, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 119526885.0, "reward": 0.54296875, "reward_std": 0.2985538840293884, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 263, "tools/generated_tokens": 4239.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1385.0703125, "completions/mean_terminated_length": 1258.651123046875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2646393794566393, "epoch": 0.04498690012141351, "frac_reward_zero_std": 0.5, "grad_norm": 0.18296407163143158, "learning_rate": 1e-06, "loss": -0.0112, "num_tokens": 119950071.0, "reward": 0.44140625, "reward_std": 0.18880821764469147, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 264, "tools/generated_tokens": 4089.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1339.57421875, "completions/mean_terminated_length": 1158.9951171875, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.2932362789288163, "epoch": 0.045157305046115835, "frac_reward_zero_std": 0.375, "grad_norm": 0.1839541345834732, "learning_rate": 1e-06, "loss": 0.0007, "num_tokens": 120377594.0, "reward": 0.4375, "reward_std": 0.24978771805763245, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 265, "tools/generated_tokens": 5059.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.81640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1197.7109375, "completions/mean_terminated_length": 1109.75, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.31323915906250477, "epoch": 0.045327709970818156, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1793917864561081, "learning_rate": 1e-06, "loss": 0.0171, "num_tokens": 120757472.0, "reward": 0.58984375, "reward_std": 0.2542063593864441, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 266, "tools/generated_tokens": 3981.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1363.0703125, "completions/mean_terminated_length": 1110.34228515625, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.32600368186831474, "epoch": 0.04549811489552048, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1349548101425171, "learning_rate": 1e-06, "loss": 0.0239, "num_tokens": 121189682.0, "reward": 0.33203125, "reward_std": 0.16181382536888123, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 267, "tools/generated_tokens": 4915.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1355.92578125, "completions/mean_terminated_length": 1170.915771484375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.30558702535927296, "epoch": 0.045668519820222804, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17803844809532166, "learning_rate": 1e-06, "loss": 0.0096, "num_tokens": 121616399.0, "reward": 0.5234375, "reward_std": 0.2926844358444214, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 268, "tools/generated_tokens": 4883.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2019.0, "completions/mean_length": 1369.9765625, "completions/mean_terminated_length": 1175.768798828125, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.30885729752480984, "epoch": 0.04583892474492513, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1312689632177353, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 122049817.0, "reward": 0.4609375, "reward_std": 0.17436380684375763, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 269, "tools/generated_tokens": 4601.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1314.4375, "completions/mean_terminated_length": 1140.797119140625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2693759361281991, "epoch": 0.04600932966962745, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15356102585792542, "learning_rate": 1e-06, "loss": 0.0375, "num_tokens": 122466985.0, "reward": 0.60546875, "reward_std": 0.2802300453186035, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 270, "tools/generated_tokens": 4458.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1276.453125, "completions/mean_terminated_length": 1050.4444580078125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.34354778937995434, "epoch": 0.04617973459432977, "frac_reward_zero_std": 0.25, "grad_norm": 0.19494813680648804, "learning_rate": 1e-06, "loss": -0.002, "num_tokens": 122878509.0, "reward": 0.5078125, "reward_std": 0.26198214292526245, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 271, "tools/generated_tokens": 4668.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 1996.0, "completions/mean_length": 1332.9453125, "completions/mean_terminated_length": 1094.59375, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.30806681886315346, "epoch": 0.0463501395190321, "frac_reward_zero_std": 0.625, "grad_norm": 0.10586902499198914, "learning_rate": 1e-06, "loss": 0.0118, "num_tokens": 123310511.0, "reward": 0.265625, "reward_std": 0.11840169876813889, "rewards/simpleverify_reward/mean": 0.265625, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 272, "tools/generated_tokens": 4540.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2009.0, "completions/mean_length": 1262.984375, "completions/mean_terminated_length": 1108.9158935546875, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.29315103963017464, "epoch": 0.04652054444373442, "frac_reward_zero_std": 0.5, "grad_norm": 0.15973858535289764, "learning_rate": 1e-06, "loss": 0.0439, "num_tokens": 123710187.0, "reward": 0.50390625, "reward_std": 0.1848640739917755, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 273, "tools/generated_tokens": 4246.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1299.25, "completions/mean_terminated_length": 1156.465087890625, "completions/min_length": 367.0, "completions/min_terminated_length": 367.0, "entropy": 0.294969892129302, "epoch": 0.04669094936843675, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16467413306236267, "learning_rate": 1e-06, "loss": 0.019, "num_tokens": 124134523.0, "reward": 0.6484375, "reward_std": 0.3220454454421997, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 274, "tools/generated_tokens": 4451.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1439.390625, "completions/mean_terminated_length": 1298.9423828125, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.2964180205017328, "epoch": 0.04686135429313907, "frac_reward_zero_std": 0.25, "grad_norm": 0.18400876224040985, "learning_rate": 1e-06, "loss": -0.002, "num_tokens": 124580303.0, "reward": 0.390625, "reward_std": 0.29364442825317383, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 275, "tools/generated_tokens": 4655.40625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1279.59765625, "completions/mean_terminated_length": 1203.746826171875, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.24685709085315466, "epoch": 0.0470317592178414, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11953554302453995, "learning_rate": 1e-06, "loss": -0.0009, "num_tokens": 124993064.0, "reward": 0.4765625, "reward_std": 0.17396602034568787, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 276, "tools/generated_tokens": 3983.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1259.859375, "completions/mean_terminated_length": 1077.9808349609375, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.3116687685251236, "epoch": 0.04720216414254372, "frac_reward_zero_std": 0.375, "grad_norm": 0.1639777272939682, "learning_rate": 1e-06, "loss": -0.0014, "num_tokens": 125393876.0, "reward": 0.5, "reward_std": 0.19189241528511047, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 277, "tools/generated_tokens": 4099.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1145.62890625, "completions/mean_terminated_length": 1034.8114013671875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.31531943939626217, "epoch": 0.047372569067246045, "frac_reward_zero_std": 0.25, "grad_norm": 0.19424794614315033, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 125762085.0, "reward": 0.46484375, "reward_std": 0.2993527054786682, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 278, "tools/generated_tokens": 3817.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1205.4296875, "completions/mean_terminated_length": 1110.1826171875, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.305449353531003, "epoch": 0.047542973991948366, "frac_reward_zero_std": 0.5, "grad_norm": 0.14293110370635986, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 126147811.0, "reward": 0.56640625, "reward_std": 0.18771302700042725, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 279, "tools/generated_tokens": 3885.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1472.8828125, "completions/mean_terminated_length": 1269.0052490234375, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.31037183478474617, "epoch": 0.047713378916650694, "frac_reward_zero_std": 0.375, "grad_norm": 0.1498546600341797, "learning_rate": 1e-06, "loss": 0.0302, "num_tokens": 126616677.0, "reward": 0.32421875, "reward_std": 0.2544988691806793, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 280, "tools/generated_tokens": 5512.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1246.5390625, "completions/mean_terminated_length": 1127.937255859375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2865572739392519, "epoch": 0.047883783841353014, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16287490725517273, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 127020431.0, "reward": 0.63671875, "reward_std": 0.2679290771484375, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 281, "tools/generated_tokens": 4134.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1338.421875, "completions/mean_terminated_length": 1166.1942138671875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.34245736710727215, "epoch": 0.04805418876605534, "frac_reward_zero_std": 0.5, "grad_norm": 0.1597912758588791, "learning_rate": 1e-06, "loss": 0.022, "num_tokens": 127439963.0, "reward": 0.3125, "reward_std": 0.20146670937538147, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 282, "tools/generated_tokens": 4362.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1383.9453125, "completions/mean_terminated_length": 1198.0150146484375, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.3314568540081382, "epoch": 0.04822459369075766, "frac_reward_zero_std": 0.4375, "grad_norm": 0.158244788646698, "learning_rate": 1e-06, "loss": 0.0546, "num_tokens": 127886029.0, "reward": 0.32421875, "reward_std": 0.21116769313812256, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 283, "tools/generated_tokens": 4943.96875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1267.23046875, "completions/mean_terminated_length": 1175.1746826171875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.271186844445765, "epoch": 0.04839499861545999, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16395214200019836, "learning_rate": 1e-06, "loss": -0.013, "num_tokens": 128295752.0, "reward": 0.62109375, "reward_std": 0.28898242115974426, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 284, "tools/generated_tokens": 4019.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1187.234375, "completions/mean_terminated_length": 1018.2990112304688, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.31557429023087025, "epoch": 0.04856540354016231, "frac_reward_zero_std": 0.375, "grad_norm": 0.1610065996646881, "learning_rate": 1e-06, "loss": 0.0314, "num_tokens": 128672052.0, "reward": 0.48828125, "reward_std": 0.2523331046104431, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 285, "tools/generated_tokens": 4123.23828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1420.62109375, "completions/mean_terminated_length": 1155.727783203125, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.28226154297590256, "epoch": 0.04873580846486463, "frac_reward_zero_std": 0.375, "grad_norm": 0.14174875617027283, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 129126259.0, "reward": 0.37890625, "reward_std": 0.2175418734550476, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 286, "tools/generated_tokens": 5596.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1163.99609375, "completions/mean_terminated_length": 1080.8846435546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2925511756911874, "epoch": 0.04890621338956696, "frac_reward_zero_std": 0.1875, "grad_norm": 0.180791974067688, "learning_rate": 1e-06, "loss": 0.0197, "num_tokens": 129510738.0, "reward": 0.62109375, "reward_std": 0.28455185890197754, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 287, "tools/generated_tokens": 4236.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2000.0, "completions/mean_length": 1299.37890625, "completions/mean_terminated_length": 1192.43310546875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2576394444331527, "epoch": 0.04907661831426928, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13466109335422516, "learning_rate": 1e-06, "loss": 0.0274, "num_tokens": 129930083.0, "reward": 0.6484375, "reward_std": 0.20960845053195953, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 288, "tools/generated_tokens": 3923.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1504.68359375, "completions/mean_terminated_length": 1156.423095703125, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.34600187093019485, "epoch": 0.04924702323897161, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11934173852205276, "learning_rate": 1e-06, "loss": 0.0333, "num_tokens": 130399650.0, "reward": 0.375, "reward_std": 0.19156451523303986, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 289, "tools/generated_tokens": 5672.703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.03515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1407.26953125, "completions/mean_terminated_length": 1116.0341796875, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.2882102522999048, "epoch": 0.04941742816367393, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1570221483707428, "learning_rate": 1e-06, "loss": 0.0354, "num_tokens": 130845255.0, "reward": 0.41796875, "reward_std": 0.2548314929008484, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 290, "tools/generated_tokens": 5023.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1287.3671875, "completions/mean_terminated_length": 1098.1365966796875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2894864585250616, "epoch": 0.049587833088376256, "frac_reward_zero_std": 0.375, "grad_norm": 0.14320386946201324, "learning_rate": 1e-06, "loss": 0.0227, "num_tokens": 131256117.0, "reward": 0.5546875, "reward_std": 0.2568049728870392, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 291, "tools/generated_tokens": 4663.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1262.24609375, "completions/mean_terminated_length": 1133.6680908203125, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.3162839636206627, "epoch": 0.049758238013078576, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17364174127578735, "learning_rate": 1e-06, "loss": 0.0313, "num_tokens": 131659396.0, "reward": 0.27734375, "reward_std": 0.2902497947216034, "rewards/simpleverify_reward/mean": 0.27734375, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 292, "tools/generated_tokens": 4454.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1156.20703125, "completions/mean_terminated_length": 1037.827392578125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.29122194834053516, "epoch": 0.049928642937780904, "frac_reward_zero_std": 0.0625, "grad_norm": 0.22128801047801971, "learning_rate": 1e-06, "loss": -0.0045, "num_tokens": 132040713.0, "reward": 0.48046875, "reward_std": 0.3702397346496582, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 293, "tools/generated_tokens": 4332.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1352.2890625, "completions/mean_terminated_length": 1215.7523193359375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2850738409906626, "epoch": 0.050099047862483224, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1403249055147171, "learning_rate": 1e-06, "loss": -0.0267, "num_tokens": 132463282.0, "reward": 0.51953125, "reward_std": 0.2100876271724701, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 294, "tools/generated_tokens": 4352.3046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1317.55859375, "completions/mean_terminated_length": 1174.200927734375, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2910716813057661, "epoch": 0.05026945278718555, "frac_reward_zero_std": 0.375, "grad_norm": 0.16851194202899933, "learning_rate": 1e-06, "loss": -0.0114, "num_tokens": 132877361.0, "reward": 0.51171875, "reward_std": 0.28488922119140625, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 295, "tools/generated_tokens": 4637.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1414.515625, "completions/mean_terminated_length": 1207.7305908203125, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.30923354625701904, "epoch": 0.05043985771188787, "frac_reward_zero_std": 0.375, "grad_norm": 0.14795434474945068, "learning_rate": 1e-06, "loss": -0.006, "num_tokens": 133320821.0, "reward": 0.4375, "reward_std": 0.22468777000904083, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 296, "tools/generated_tokens": 4742.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1341.859375, "completions/mean_terminated_length": 1166.190185546875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.3143516555428505, "epoch": 0.0506102626365902, "frac_reward_zero_std": 0.25, "grad_norm": 0.16825224459171295, "learning_rate": 1e-06, "loss": -0.01, "num_tokens": 133743697.0, "reward": 0.53125, "reward_std": 0.2893039882183075, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 297, "tools/generated_tokens": 4789.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1227.21875, "completions/mean_terminated_length": 1079.705078125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.25937829725444317, "epoch": 0.05078066756129252, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1845153272151947, "learning_rate": 1e-06, "loss": 0.0043, "num_tokens": 134141257.0, "reward": 0.47265625, "reward_std": 0.21003374457359314, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 298, "tools/generated_tokens": 4475.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1175.48046875, "completions/mean_terminated_length": 1041.851318359375, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.2886015884578228, "epoch": 0.05095107248599485, "frac_reward_zero_std": 0.5, "grad_norm": 0.13403676450252533, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 134514916.0, "reward": 0.60546875, "reward_std": 0.2085040807723999, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 299, "tools/generated_tokens": 3647.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1383.234375, "completions/mean_terminated_length": 1161.6458740234375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.32432376593351364, "epoch": 0.05112147741069717, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17836932837963104, "learning_rate": 1e-06, "loss": 0.0567, "num_tokens": 134950336.0, "reward": 0.46484375, "reward_std": 0.31528323888778687, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 300, "tools/generated_tokens": 5023.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1390.16015625, "completions/mean_terminated_length": 1170.8802490234375, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.28654346987605095, "epoch": 0.05129188233539949, "frac_reward_zero_std": 0.4375, "grad_norm": 0.22838200628757477, "learning_rate": 1e-06, "loss": 0.0012, "num_tokens": 135382921.0, "reward": 0.3125, "reward_std": 0.20687922835350037, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 301, "tools/generated_tokens": 4806.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1296.2890625, "completions/mean_terminated_length": 1090.6019287109375, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.32894255965948105, "epoch": 0.05146228726010182, "frac_reward_zero_std": 0.5, "grad_norm": 0.14717616140842438, "learning_rate": 1e-06, "loss": 0.0265, "num_tokens": 135794131.0, "reward": 0.44140625, "reward_std": 0.21863040328025818, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 302, "tools/generated_tokens": 4944.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1415.41796875, "completions/mean_terminated_length": 1191.174560546875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.3031544340774417, "epoch": 0.05163269218480414, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20804932713508606, "learning_rate": 1e-06, "loss": -0.008, "num_tokens": 136254958.0, "reward": 0.375, "reward_std": 0.24233347177505493, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 303, "tools/generated_tokens": 4863.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1237.66796875, "completions/mean_terminated_length": 1187.232421875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.26903535425662994, "epoch": 0.051803097109506466, "frac_reward_zero_std": 0.375, "grad_norm": 0.16161197423934937, "learning_rate": 1e-06, "loss": 0.0379, "num_tokens": 136645689.0, "reward": 0.70703125, "reward_std": 0.2602487802505493, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 304, "tools/generated_tokens": 3581.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2014.0, "completions/mean_length": 1308.9765625, "completions/mean_terminated_length": 1232.52587890625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.3003286551684141, "epoch": 0.051973502034208786, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16468428075313568, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 137054851.0, "reward": 0.59765625, "reward_std": 0.2183406949043274, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 305, "tools/generated_tokens": 4068.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1354.546875, "completions/mean_terminated_length": 1113.66845703125, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.2759744944050908, "epoch": 0.052143906958911114, "frac_reward_zero_std": 0.375, "grad_norm": 0.1579194813966751, "learning_rate": 1e-06, "loss": 0.0297, "num_tokens": 137490847.0, "reward": 0.375, "reward_std": 0.256390780210495, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 306, "tools/generated_tokens": 5098.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1155.61328125, "completions/mean_terminated_length": 1063.29736328125, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.29237478971481323, "epoch": 0.052314311883613435, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1777174174785614, "learning_rate": 1e-06, "loss": 0.0371, "num_tokens": 137862716.0, "reward": 0.64453125, "reward_std": 0.2609933912754059, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 307, "tools/generated_tokens": 3939.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1302.82421875, "completions/mean_terminated_length": 1064.685546875, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.28632466681301594, "epoch": 0.05248471680831576, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16957347095012665, "learning_rate": 1e-06, "loss": 0.0191, "num_tokens": 138279983.0, "reward": 0.50390625, "reward_std": 0.2303449958562851, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 308, "tools/generated_tokens": 4750.83984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1420.8125, "completions/mean_terminated_length": 1170.6337890625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.28520943596959114, "epoch": 0.05265512173301808, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1310110241174698, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 138725119.0, "reward": 0.45703125, "reward_std": 0.23404711484909058, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 309, "tools/generated_tokens": 5196.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1313.05078125, "completions/mean_terminated_length": 1125.7156982421875, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.36094664968550205, "epoch": 0.05282552665772041, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18607859313488007, "learning_rate": 1e-06, "loss": 0.0292, "num_tokens": 139150236.0, "reward": 0.44921875, "reward_std": 0.245716854929924, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 310, "tools/generated_tokens": 4705.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1431.7578125, "completions/mean_terminated_length": 1190.625, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.339433029294014, "epoch": 0.05299593158242273, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13819488883018494, "learning_rate": 1e-06, "loss": 0.0466, "num_tokens": 139589838.0, "reward": 0.37890625, "reward_std": 0.20388561487197876, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 311, "tools/generated_tokens": 4759.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1355.0546875, "completions/mean_terminated_length": 1124.078125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.28534994274377823, "epoch": 0.05316633650712506, "frac_reward_zero_std": 0.375, "grad_norm": 0.16902461647987366, "learning_rate": 1e-06, "loss": 0.0052, "num_tokens": 140019132.0, "reward": 0.44140625, "reward_std": 0.26452332735061646, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 312, "tools/generated_tokens": 4787.06640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.67578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1153.1875, "completions/mean_terminated_length": 1064.8626708984375, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.34434532187879086, "epoch": 0.05333674143182738, "frac_reward_zero_std": 0.375, "grad_norm": 0.17630203068256378, "learning_rate": 1e-06, "loss": -0.0073, "num_tokens": 140396508.0, "reward": 0.4609375, "reward_std": 0.23819956183433533, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 313, "tools/generated_tokens": 4025.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1354.375, "completions/mean_terminated_length": 1137.3948974609375, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "entropy": 0.32829746417701244, "epoch": 0.05350714635652971, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17183540761470795, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 140831036.0, "reward": 0.36328125, "reward_std": 0.3206353783607483, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 314, "tools/generated_tokens": 5274.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1262.4140625, "completions/mean_terminated_length": 1142.09912109375, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.2923651207238436, "epoch": 0.05367755128123203, "frac_reward_zero_std": 0.25, "grad_norm": 0.17488974332809448, "learning_rate": 1e-06, "loss": 0.0046, "num_tokens": 141240950.0, "reward": 0.50390625, "reward_std": 0.326728880405426, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 315, "tools/generated_tokens": 4366.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1423.34765625, "completions/mean_terminated_length": 1188.263427734375, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.27633984480053186, "epoch": 0.05384795620593435, "frac_reward_zero_std": 0.5, "grad_norm": 0.12932582199573517, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 141686511.0, "reward": 0.328125, "reward_std": 0.19846853613853455, "rewards/simpleverify_reward/mean": 0.328125, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 316, "tools/generated_tokens": 5119.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1349.7265625, "completions/mean_terminated_length": 1167.4285888671875, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.29339468479156494, "epoch": 0.054018361130636676, "frac_reward_zero_std": 0.125, "grad_norm": 0.18806594610214233, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 142120489.0, "reward": 0.4375, "reward_std": 0.3388923406600952, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 317, "tools/generated_tokens": 4981.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1343.5078125, "completions/mean_terminated_length": 1256.9912109375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.26533154770731926, "epoch": 0.054188766055339, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12821224331855774, "learning_rate": 1e-06, "loss": 0.0102, "num_tokens": 142529371.0, "reward": 0.50390625, "reward_std": 0.15613234043121338, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 318, "tools/generated_tokens": 3607.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1137.10546875, "completions/mean_terminated_length": 1011.6044921875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.2547017401084304, "epoch": 0.054359170980041324, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1771804690361023, "learning_rate": 1e-06, "loss": 0.0389, "num_tokens": 142895526.0, "reward": 0.609375, "reward_std": 0.23283424973487854, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 319, "tools/generated_tokens": 3753.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1260.55078125, "completions/mean_terminated_length": 1135.8416748046875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.27737378515303135, "epoch": 0.054529575904743645, "frac_reward_zero_std": 0.5, "grad_norm": 0.1308823972940445, "learning_rate": 1e-06, "loss": 0.0094, "num_tokens": 143295379.0, "reward": 0.3984375, "reward_std": 0.16691282391548157, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 320, "tools/generated_tokens": 4052.55859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1292.5546875, "completions/mean_terminated_length": 1217.9827880859375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.29106081649661064, "epoch": 0.05469998082944597, "frac_reward_zero_std": 0.125, "grad_norm": 0.1829807162284851, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 143706257.0, "reward": 0.40234375, "reward_std": 0.36748284101486206, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 321, "tools/generated_tokens": 4348.55859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1392.87890625, "completions/mean_terminated_length": 1241.6971435546875, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.29241783916950226, "epoch": 0.05487038575414829, "frac_reward_zero_std": 0.25, "grad_norm": 0.1687900274991989, "learning_rate": 1e-06, "loss": 0.0421, "num_tokens": 144143938.0, "reward": 0.46875, "reward_std": 0.29602646827697754, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 322, "tools/generated_tokens": 4840.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1342.921875, "completions/mean_terminated_length": 1158.84228515625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.3193067070096731, "epoch": 0.05504079067885062, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19132868945598602, "learning_rate": 1e-06, "loss": 0.017, "num_tokens": 144585150.0, "reward": 0.4296875, "reward_std": 0.2655054032802582, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 323, "tools/generated_tokens": 4814.94140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1299.42578125, "completions/mean_terminated_length": 1203.7928466796875, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.3053978104144335, "epoch": 0.05521119560355294, "frac_reward_zero_std": 0.375, "grad_norm": 0.16768132150173187, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 144995419.0, "reward": 0.34765625, "reward_std": 0.24562345445156097, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 324, "tools/generated_tokens": 4115.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1207.55078125, "completions/mean_terminated_length": 1083.1839599609375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.31183927692472935, "epoch": 0.05538160052825527, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17047731578350067, "learning_rate": 1e-06, "loss": 0.0214, "num_tokens": 145393080.0, "reward": 0.63671875, "reward_std": 0.27880415320396423, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 325, "tools/generated_tokens": 4255.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1242.26171875, "completions/mean_terminated_length": 1162.73388671875, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.2630101628601551, "epoch": 0.05555200545295759, "frac_reward_zero_std": 0.5, "grad_norm": 0.14168424904346466, "learning_rate": 1e-06, "loss": 0.0229, "num_tokens": 145800203.0, "reward": 0.6171875, "reward_std": 0.20215418934822083, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 326, "tools/generated_tokens": 3938.27734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1313.8046875, "completions/mean_terminated_length": 1063.952880859375, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.35248119942843914, "epoch": 0.05572241037765992, "frac_reward_zero_std": 0.375, "grad_norm": 0.1601688712835312, "learning_rate": 1e-06, "loss": 0.0226, "num_tokens": 146218825.0, "reward": 0.35546875, "reward_std": 0.24520954489707947, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 327, "tools/generated_tokens": 4833.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1230.7734375, "completions/mean_terminated_length": 1097.04541015625, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.30171683616936207, "epoch": 0.05589281530236224, "frac_reward_zero_std": 0.375, "grad_norm": 0.16057145595550537, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 146616991.0, "reward": 0.39453125, "reward_std": 0.22226692736148834, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 328, "tools/generated_tokens": 4422.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1200.1953125, "completions/mean_terminated_length": 1143.675048828125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.284699235111475, "epoch": 0.056063220227064565, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15954619646072388, "learning_rate": 1e-06, "loss": -0.0055, "num_tokens": 147005265.0, "reward": 0.58203125, "reward_std": 0.2673723101615906, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 329, "tools/generated_tokens": 3824.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1379.62890625, "completions/mean_terminated_length": 1252.172119140625, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.3053628709167242, "epoch": 0.056233625151766886, "frac_reward_zero_std": 0.25, "grad_norm": 0.16071152687072754, "learning_rate": 1e-06, "loss": 0.0109, "num_tokens": 147438146.0, "reward": 0.609375, "reward_std": 0.304276704788208, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 330, "tools/generated_tokens": 4427.625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.36328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2011.0, "completions/mean_length": 1453.5, "completions/mean_terminated_length": 1114.343505859375, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.34157600067555904, "epoch": 0.05640403007646921, "frac_reward_zero_std": 0.5, "grad_norm": 0.13115476071834564, "learning_rate": 1e-06, "loss": 0.0216, "num_tokens": 147895138.0, "reward": 0.3671875, "reward_std": 0.2007330358028412, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 331, "tools/generated_tokens": 5541.52734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1347.921875, "completions/mean_terminated_length": 1094.718017578125, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.31185402534902096, "epoch": 0.056574435001171534, "frac_reward_zero_std": 0.375, "grad_norm": 0.16882377862930298, "learning_rate": 1e-06, "loss": 0.031, "num_tokens": 148331582.0, "reward": 0.265625, "reward_std": 0.18420085310935974, "rewards/simpleverify_reward/mean": 0.265625, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 332, "tools/generated_tokens": 5195.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1307.73046875, "completions/mean_terminated_length": 1109.836669921875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.29180445708334446, "epoch": 0.056744839925873855, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17275694012641907, "learning_rate": 1e-06, "loss": 0.0276, "num_tokens": 148750601.0, "reward": 0.53515625, "reward_std": 0.2933111786842346, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 333, "tools/generated_tokens": 4939.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1329.640625, "completions/mean_terminated_length": 1176.4407958984375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2849424909800291, "epoch": 0.05691524485057618, "frac_reward_zero_std": 0.375, "grad_norm": 0.1704791635274887, "learning_rate": 1e-06, "loss": 0.0101, "num_tokens": 149177085.0, "reward": 0.4921875, "reward_std": 0.25355497002601624, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 334, "tools/generated_tokens": 4553.65234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1286.7890625, "completions/mean_terminated_length": 1038.3211669921875, "completions/min_length": 405.0, "completions/min_terminated_length": 405.0, "entropy": 0.33647651597857475, "epoch": 0.0570856497752785, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1627286821603775, "learning_rate": 1e-06, "loss": 0.0078, "num_tokens": 149586551.0, "reward": 0.26171875, "reward_std": 0.22039085626602173, "rewards/simpleverify_reward/mean": 0.26171875, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 335, "tools/generated_tokens": 4806.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1992.0, "completions/mean_length": 1180.13671875, "completions/mean_terminated_length": 1056.15625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.2707588989287615, "epoch": 0.05725605469998083, "frac_reward_zero_std": 0.1875, "grad_norm": 0.20591795444488525, "learning_rate": 1e-06, "loss": 0.0251, "num_tokens": 149969866.0, "reward": 0.5, "reward_std": 0.30712568759918213, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 336, "tools/generated_tokens": 4036.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1382.7890625, "completions/mean_terminated_length": 1209.1280517578125, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.31500553060323, "epoch": 0.05742645962468315, "frac_reward_zero_std": 0.625, "grad_norm": 0.09929162263870239, "learning_rate": 1e-06, "loss": -0.0164, "num_tokens": 150400244.0, "reward": 0.46484375, "reward_std": 0.11039985716342926, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 337, "tools/generated_tokens": 4390.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.27734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1521.7578125, "completions/mean_terminated_length": 1319.810791015625, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.2835215609520674, "epoch": 0.05759686454938548, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11700831353664398, "learning_rate": 1e-06, "loss": 0.0187, "num_tokens": 150866198.0, "reward": 0.46875, "reward_std": 0.16707327961921692, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 338, "tools/generated_tokens": 4929.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1209.3984375, "completions/mean_terminated_length": 1072.1727294921875, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.2663586363196373, "epoch": 0.0577672694740878, "frac_reward_zero_std": 0.375, "grad_norm": 0.16561011970043182, "learning_rate": 1e-06, "loss": 0.0185, "num_tokens": 151251996.0, "reward": 0.5234375, "reward_std": 0.2843548059463501, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 339, "tools/generated_tokens": 4169.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1271.2734375, "completions/mean_terminated_length": 1160.3125, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.28707977943122387, "epoch": 0.05793767439879013, "frac_reward_zero_std": 0.625, "grad_norm": 0.12584614753723145, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 151667730.0, "reward": 0.5703125, "reward_std": 0.13149452209472656, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 340, "tools/generated_tokens": 3807.27734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1407.93359375, "completions/mean_terminated_length": 1244.7843017578125, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.25658425129950047, "epoch": 0.05810807932349245, "frac_reward_zero_std": 0.5, "grad_norm": 0.11719018220901489, "learning_rate": 1e-06, "loss": 0.0047, "num_tokens": 152114033.0, "reward": 0.49609375, "reward_std": 0.15834102034568787, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 341, "tools/generated_tokens": 4631.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1293.5234375, "completions/mean_terminated_length": 1145.4532470703125, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.3050544150173664, "epoch": 0.058278484248194776, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1705986112356186, "learning_rate": 1e-06, "loss": 0.0077, "num_tokens": 152524375.0, "reward": 0.265625, "reward_std": 0.2792971134185791, "rewards/simpleverify_reward/mean": 0.265625, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 342, "tools/generated_tokens": 4125.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1317.09375, "completions/mean_terminated_length": 1152.7320556640625, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.2780300956219435, "epoch": 0.058448889172897096, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17361460626125336, "learning_rate": 1e-06, "loss": 0.0221, "num_tokens": 152937135.0, "reward": 0.578125, "reward_std": 0.30502164363861084, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 343, "tools/generated_tokens": 4517.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1232.47265625, "completions/mean_terminated_length": 1090.31640625, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2694319849833846, "epoch": 0.058619294097599424, "frac_reward_zero_std": 0.1875, "grad_norm": 0.2018771916627884, "learning_rate": 1e-06, "loss": 0.0195, "num_tokens": 153338744.0, "reward": 0.6171875, "reward_std": 0.3160597085952759, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 344, "tools/generated_tokens": 4336.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1309.08203125, "completions/mean_terminated_length": 1180.27978515625, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.3142383638769388, "epoch": 0.058789699022301745, "frac_reward_zero_std": 0.375, "grad_norm": 0.16071529686450958, "learning_rate": 1e-06, "loss": 0.0328, "num_tokens": 153758045.0, "reward": 0.55078125, "reward_std": 0.27139341831207275, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 345, "tools/generated_tokens": 4269.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1481.47265625, "completions/mean_terminated_length": 1194.8883056640625, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.2590813608840108, "epoch": 0.058960103947004065, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13019989430904388, "learning_rate": 1e-06, "loss": 0.0193, "num_tokens": 154212246.0, "reward": 0.390625, "reward_std": 0.2226376086473465, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 346, "tools/generated_tokens": 5049.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1335.6171875, "completions/mean_terminated_length": 1265.296142578125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.28239644318819046, "epoch": 0.05913050887170639, "frac_reward_zero_std": 0.625, "grad_norm": 0.13273970782756805, "learning_rate": 1e-06, "loss": 0.0063, "num_tokens": 154631492.0, "reward": 0.515625, "reward_std": 0.12136821448802948, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 347, "tools/generated_tokens": 4191.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1313.88671875, "completions/mean_terminated_length": 1216.4556884765625, "completions/min_length": 301.0, "completions/min_terminated_length": 301.0, "entropy": 0.290899645537138, "epoch": 0.059300913796408714, "frac_reward_zero_std": 0.25, "grad_norm": 0.17403538525104523, "learning_rate": 1e-06, "loss": 0.0375, "num_tokens": 155045191.0, "reward": 0.546875, "reward_std": 0.26983416080474854, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 348, "tools/generated_tokens": 4561.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1295.77734375, "completions/mean_terminated_length": 1210.743408203125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.29990064818412066, "epoch": 0.05947131872111104, "frac_reward_zero_std": 0.25, "grad_norm": 0.17534621059894562, "learning_rate": 1e-06, "loss": 0.0128, "num_tokens": 155456238.0, "reward": 0.5234375, "reward_std": 0.29847443103790283, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 349, "tools/generated_tokens": 4311.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1355.2109375, "completions/mean_terminated_length": 1252.690673828125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2619368303567171, "epoch": 0.05964172364581336, "frac_reward_zero_std": 0.625, "grad_norm": 0.10621669888496399, "learning_rate": 1e-06, "loss": -0.0139, "num_tokens": 155880852.0, "reward": 0.4453125, "reward_std": 0.14954319596290588, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 350, "tools/generated_tokens": 4259.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1201.28125, "completions/mean_terminated_length": 1105.5694580078125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2944045700132847, "epoch": 0.05981212857051569, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1531706303358078, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 156275180.0, "reward": 0.61328125, "reward_std": 0.2757830023765564, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 351, "tools/generated_tokens": 4137.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1337.43359375, "completions/mean_terminated_length": 1224.905029296875, "completions/min_length": 211.0, "completions/min_terminated_length": 211.0, "entropy": 0.26592031866312027, "epoch": 0.05998253349521801, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14751036465168, "learning_rate": 1e-06, "loss": -0.009, "num_tokens": 156692843.0, "reward": 0.42578125, "reward_std": 0.22358623147010803, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 352, "tools/generated_tokens": 4361.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1292.46484375, "completions/mean_terminated_length": 1085.726318359375, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.27806926518678665, "epoch": 0.06015293841992034, "frac_reward_zero_std": 0.5, "grad_norm": 0.13071857392787933, "learning_rate": 1e-06, "loss": 0.0119, "num_tokens": 157107282.0, "reward": 0.328125, "reward_std": 0.19673973321914673, "rewards/simpleverify_reward/mean": 0.328125, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 353, "tools/generated_tokens": 4452.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1312.4921875, "completions/mean_terminated_length": 1184.2843017578125, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.25763664301484823, "epoch": 0.06032334334462266, "frac_reward_zero_std": 0.375, "grad_norm": 0.14546117186546326, "learning_rate": 1e-06, "loss": 0.0232, "num_tokens": 157524192.0, "reward": 0.515625, "reward_std": 0.21455954015254974, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 354, "tools/generated_tokens": 4216.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1212.2578125, "completions/mean_terminated_length": 1101.318603515625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2963197957724333, "epoch": 0.060493748269324986, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2035531848669052, "learning_rate": 1e-06, "loss": 0.0329, "num_tokens": 157915042.0, "reward": 0.51953125, "reward_std": 0.30379754304885864, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 355, "tools/generated_tokens": 4100.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1256.3203125, "completions/mean_terminated_length": 1203.541748046875, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.2689771419391036, "epoch": 0.06066415319402731, "frac_reward_zero_std": 0.375, "grad_norm": 0.1614867001771927, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 158314004.0, "reward": 0.6015625, "reward_std": 0.25890904664993286, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 356, "tools/generated_tokens": 3888.33203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2007.0, "completions/mean_length": 1229.04296875, "completions/mean_terminated_length": 1128.4736328125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.300431115552783, "epoch": 0.060834558118729634, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15670716762542725, "learning_rate": 1e-06, "loss": 0.0242, "num_tokens": 158710735.0, "reward": 0.51953125, "reward_std": 0.19652670621871948, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 357, "tools/generated_tokens": 4037.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1245.8515625, "completions/mean_terminated_length": 1166.6695556640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2689073383808136, "epoch": 0.061004963043431955, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17072314023971558, "learning_rate": 1e-06, "loss": 0.0399, "num_tokens": 159103593.0, "reward": 0.67578125, "reward_std": 0.22358150780200958, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 358, "tools/generated_tokens": 3533.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1285.4375, "completions/mean_terminated_length": 1113.9521484375, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.29637575056403875, "epoch": 0.06117536796813428, "frac_reward_zero_std": 0.375, "grad_norm": 0.16646058857440948, "learning_rate": 1e-06, "loss": -0.0002, "num_tokens": 159513913.0, "reward": 0.45703125, "reward_std": 0.24538421630859375, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 359, "tools/generated_tokens": 4637.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.63671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1215.859375, "completions/mean_terminated_length": 1079.69091796875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.2793931197375059, "epoch": 0.0613457728928366, "frac_reward_zero_std": 0.5, "grad_norm": 0.15953166782855988, "learning_rate": 1e-06, "loss": 0.0144, "num_tokens": 159898293.0, "reward": 0.5, "reward_std": 0.21438735723495483, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 360, "tools/generated_tokens": 3711.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1542.06640625, "completions/mean_terminated_length": 1217.769287109375, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "entropy": 0.2827363107353449, "epoch": 0.061516177817538924, "frac_reward_zero_std": 0.5, "grad_norm": 0.16723506152629852, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 160374646.0, "reward": 0.30859375, "reward_std": 0.161190003156662, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 361, "tools/generated_tokens": 5526.1015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1998.0, "completions/mean_length": 1189.5234375, "completions/mean_terminated_length": 1092.478271484375, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.27755394764244556, "epoch": 0.06168658274224125, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20288144052028656, "learning_rate": 1e-06, "loss": -0.0119, "num_tokens": 160754044.0, "reward": 0.640625, "reward_std": 0.23778341710567474, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 362, "tools/generated_tokens": 3629.52734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1423.7109375, "completions/mean_terminated_length": 1240.83837890625, "completions/min_length": 366.0, "completions/min_terminated_length": 366.0, "entropy": 0.3219546005129814, "epoch": 0.06185698766694357, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13976918160915375, "learning_rate": 1e-06, "loss": 0.0118, "num_tokens": 161199090.0, "reward": 0.47265625, "reward_std": 0.19825831055641174, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 363, "tools/generated_tokens": 4967.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1211.7109375, "completions/mean_terminated_length": 1109.0087890625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.3026361558586359, "epoch": 0.0620273925916459, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1752837598323822, "learning_rate": 1e-06, "loss": 0.0242, "num_tokens": 161587336.0, "reward": 0.56640625, "reward_std": 0.26411134004592896, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 364, "tools/generated_tokens": 4027.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1468.41796875, "completions/mean_terminated_length": 1175.2235107421875, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.3634508866816759, "epoch": 0.06219779751634822, "frac_reward_zero_std": 0.5, "grad_norm": 0.14967897534370422, "learning_rate": 1e-06, "loss": 0.0301, "num_tokens": 162057123.0, "reward": 0.4453125, "reward_std": 0.21996080875396729, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 365, "tools/generated_tokens": 5540.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.98828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1338.390625, "completions/mean_terminated_length": 1166.1553955078125, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.32216991670429707, "epoch": 0.06236820244105055, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1275116205215454, "learning_rate": 1e-06, "loss": 0.0251, "num_tokens": 162492103.0, "reward": 0.34765625, "reward_std": 0.18408125638961792, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 366, "tools/generated_tokens": 5074.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1334.4921875, "completions/mean_terminated_length": 1156.990234375, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "entropy": 0.30862087197601795, "epoch": 0.06253860736575287, "frac_reward_zero_std": 0.25, "grad_norm": 0.17597267031669617, "learning_rate": 1e-06, "loss": 0.0223, "num_tokens": 162922981.0, "reward": 0.43359375, "reward_std": 0.31944963335990906, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 367, "tools/generated_tokens": 4902.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1263.9453125, "completions/mean_terminated_length": 1147.923828125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.26257198210805655, "epoch": 0.0627090122904552, "frac_reward_zero_std": 0.5, "grad_norm": 0.13663767278194427, "learning_rate": 1e-06, "loss": 0.0072, "num_tokens": 163325607.0, "reward": 0.515625, "reward_std": 0.17686697840690613, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 368, "tools/generated_tokens": 3903.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1409.3984375, "completions/mean_terminated_length": 1276.867919921875, "completions/min_length": 373.0, "completions/min_terminated_length": 373.0, "entropy": 0.28452976047992706, "epoch": 0.06287941721515752, "frac_reward_zero_std": 0.1875, "grad_norm": 0.19804418087005615, "learning_rate": 1e-06, "loss": 0.0198, "num_tokens": 163768877.0, "reward": 0.578125, "reward_std": 0.3485180139541626, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 369, "tools/generated_tokens": 4833.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1232.08984375, "completions/mean_terminated_length": 1119.675537109375, "completions/min_length": 300.0, "completions/min_terminated_length": 300.0, "entropy": 0.2565639251843095, "epoch": 0.06304982213985984, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1488446742296219, "learning_rate": 1e-06, "loss": 0.0223, "num_tokens": 164175940.0, "reward": 0.53125, "reward_std": 0.24017895758152008, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 370, "tools/generated_tokens": 4232.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1324.9765625, "completions/mean_terminated_length": 1013.9552612304688, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.28396870102733374, "epoch": 0.06322022706456217, "frac_reward_zero_std": 0.375, "grad_norm": 0.1640467494726181, "learning_rate": 1e-06, "loss": 0.0338, "num_tokens": 164608078.0, "reward": 0.40234375, "reward_std": 0.23206059634685516, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 371, "tools/generated_tokens": 5060.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1427.19140625, "completions/mean_terminated_length": 1232.98974609375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.2992200646549463, "epoch": 0.06339063198926449, "frac_reward_zero_std": 0.25, "grad_norm": 0.19355005025863647, "learning_rate": 1e-06, "loss": 0.0016, "num_tokens": 165063887.0, "reward": 0.36328125, "reward_std": 0.31170564889907837, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 372, "tools/generated_tokens": 5099.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.79296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1239.25390625, "completions/mean_terminated_length": 1166.98291015625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.28434338979423046, "epoch": 0.06356103691396682, "frac_reward_zero_std": 0.125, "grad_norm": 0.17379023134708405, "learning_rate": 1e-06, "loss": 0.0155, "num_tokens": 165460608.0, "reward": 0.4609375, "reward_std": 0.31952911615371704, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 373, "tools/generated_tokens": 3975.26171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1408.8984375, "completions/mean_terminated_length": 1168.3763427734375, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.3091003466397524, "epoch": 0.06373144183866913, "frac_reward_zero_std": 0.625, "grad_norm": 0.1238866001367569, "learning_rate": 1e-06, "loss": 0.0056, "num_tokens": 165904230.0, "reward": 0.47265625, "reward_std": 0.1471242755651474, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 374, "tools/generated_tokens": 4888.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1331.87890625, "completions/mean_terminated_length": 1233.21337890625, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.3320555854588747, "epoch": 0.06390184676337146, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17904691398143768, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 166336199.0, "reward": 0.421875, "reward_std": 0.3037048578262329, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 375, "tools/generated_tokens": 5179.89453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1199.71875, "completions/mean_terminated_length": 1087.114990234375, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.26301499642431736, "epoch": 0.06407225168807379, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14704641699790955, "learning_rate": 1e-06, "loss": -0.0169, "num_tokens": 166730767.0, "reward": 0.5234375, "reward_std": 0.22964167594909668, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 376, "tools/generated_tokens": 4127.72265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1252.3125, "completions/mean_terminated_length": 1192.134521484375, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.27136948611587286, "epoch": 0.0642426566127761, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18955622613430023, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 167125375.0, "reward": 0.73046875, "reward_std": 0.22005823254585266, "rewards/simpleverify_reward/mean": 0.73046875, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 377, "tools/generated_tokens": 3732.31640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1210.23828125, "completions/mean_terminated_length": 1127.5450439453125, "completions/min_length": 299.0, "completions/min_terminated_length": 299.0, "entropy": 0.2912600552663207, "epoch": 0.06441306153747843, "frac_reward_zero_std": 0.375, "grad_norm": 0.15348312258720398, "learning_rate": 1e-06, "loss": 0.028, "num_tokens": 167525692.0, "reward": 0.53515625, "reward_std": 0.2748759984970093, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 378, "tools/generated_tokens": 4418.2421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1383.22265625, "completions/mean_terminated_length": 1166.2279052734375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.2929877061396837, "epoch": 0.06458346646218076, "frac_reward_zero_std": 0.375, "grad_norm": 0.1400548219680786, "learning_rate": 1e-06, "loss": 0.0432, "num_tokens": 167961989.0, "reward": 0.4765625, "reward_std": 0.25640395283699036, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 379, "tools/generated_tokens": 4695.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1380.73828125, "completions/mean_terminated_length": 1210.6519775390625, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.2998197767883539, "epoch": 0.06475387138688309, "frac_reward_zero_std": 0.375, "grad_norm": 0.18942537903785706, "learning_rate": 1e-06, "loss": 0.015, "num_tokens": 168400466.0, "reward": 0.3515625, "reward_std": 0.2543018162250519, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 380, "tools/generated_tokens": 4604.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1515.8515625, "completions/mean_terminated_length": 1246.6529541015625, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.2988246390596032, "epoch": 0.0649242763115854, "frac_reward_zero_std": 0.5, "grad_norm": 0.12259532511234283, "learning_rate": 1e-06, "loss": -0.0066, "num_tokens": 168874316.0, "reward": 0.25, "reward_std": 0.17693254351615906, "rewards/simpleverify_reward/mean": 0.25, "rewards/simpleverify_reward/std": 0.4338609278202057, "step": 381, "tools/generated_tokens": 5523.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1436.56640625, "completions/mean_terminated_length": 1261.43212890625, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.2791782543063164, "epoch": 0.06509468123628773, "frac_reward_zero_std": 0.375, "grad_norm": 0.1588088721036911, "learning_rate": 1e-06, "loss": 0.0342, "num_tokens": 169317021.0, "reward": 0.36328125, "reward_std": 0.26489412784576416, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 382, "tools/generated_tokens": 4196.578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1234.421875, "completions/mean_terminated_length": 1142.4521484375, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.2746760230511427, "epoch": 0.06526508616099005, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16601525247097015, "learning_rate": 1e-06, "loss": 0.0063, "num_tokens": 169703497.0, "reward": 0.6171875, "reward_std": 0.2743987441062927, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 383, "tools/generated_tokens": 3290.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.00390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1408.7109375, "completions/mean_terminated_length": 1148.7802734375, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.3064160402864218, "epoch": 0.06543549108569238, "frac_reward_zero_std": 0.5, "grad_norm": 0.1455686241388321, "learning_rate": 1e-06, "loss": 0.0308, "num_tokens": 170147087.0, "reward": 0.3359375, "reward_std": 0.18353557586669922, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 384, "tools/generated_tokens": 4928.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1340.63671875, "completions/mean_terminated_length": 1257.23583984375, "completions/min_length": 365.0, "completions/min_terminated_length": 365.0, "entropy": 0.28463104739785194, "epoch": 0.0656058960103947, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1766914278268814, "learning_rate": 1e-06, "loss": -0.0055, "num_tokens": 170572642.0, "reward": 0.5390625, "reward_std": 0.29113906621932983, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 385, "tools/generated_tokens": 3908.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1337.703125, "completions/mean_terminated_length": 1270.9273681640625, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.32302073016762733, "epoch": 0.06577630093509702, "frac_reward_zero_std": 0.375, "grad_norm": 0.15863491594791412, "learning_rate": 1e-06, "loss": -0.0072, "num_tokens": 170997558.0, "reward": 0.5546875, "reward_std": 0.2906888723373413, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 386, "tools/generated_tokens": 4369.703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1257.71484375, "completions/mean_terminated_length": 1175.9654541015625, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.2963530384004116, "epoch": 0.06594670585979935, "frac_reward_zero_std": 0.5, "grad_norm": 0.19245870411396027, "learning_rate": 1e-06, "loss": 0.0427, "num_tokens": 171408445.0, "reward": 0.4453125, "reward_std": 0.23488396406173706, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 387, "tools/generated_tokens": 4385.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1320.796875, "completions/mean_terminated_length": 1197.93603515625, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "entropy": 0.2761593796312809, "epoch": 0.06611711078450168, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3170912563800812, "learning_rate": 1e-06, "loss": 0.0046, "num_tokens": 171829161.0, "reward": 0.5, "reward_std": 0.215584397315979, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 388, "tools/generated_tokens": 4424.8203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1341.34375, "completions/mean_terminated_length": 1225.713623046875, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.27688918076455593, "epoch": 0.06628751570920399, "frac_reward_zero_std": 0.375, "grad_norm": 0.16212299466133118, "learning_rate": 1e-06, "loss": -0.0063, "num_tokens": 172251361.0, "reward": 0.4921875, "reward_std": 0.24551981687545776, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 389, "tools/generated_tokens": 4253.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1220.5625, "completions/mean_terminated_length": 1085.1680908203125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3241068311035633, "epoch": 0.06645792063390632, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18372248113155365, "learning_rate": 1e-06, "loss": 0.0169, "num_tokens": 172645409.0, "reward": 0.609375, "reward_std": 0.1969657838344574, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 390, "tools/generated_tokens": 4324.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2003.0, "completions/mean_length": 1337.53125, "completions/mean_terminated_length": 1152.0443115234375, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.3025492988526821, "epoch": 0.06662832555860865, "frac_reward_zero_std": 0.375, "grad_norm": 0.15253105759620667, "learning_rate": 1e-06, "loss": 0.0212, "num_tokens": 173075817.0, "reward": 0.44921875, "reward_std": 0.29400384426116943, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 391, "tools/generated_tokens": 5105.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.83984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1306.34765625, "completions/mean_terminated_length": 1143.8905029296875, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.2915899492800236, "epoch": 0.06679873048331096, "frac_reward_zero_std": 0.25, "grad_norm": 0.17179858684539795, "learning_rate": 1e-06, "loss": 0.0369, "num_tokens": 173500722.0, "reward": 0.42578125, "reward_std": 0.28256726264953613, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 392, "tools/generated_tokens": 4706.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1271.87890625, "completions/mean_terminated_length": 1180.3756103515625, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.31104396283626556, "epoch": 0.06696913540801329, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16390515863895416, "learning_rate": 1e-06, "loss": -0.009, "num_tokens": 173905283.0, "reward": 0.35546875, "reward_std": 0.25825291872024536, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 393, "tools/generated_tokens": 3895.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1383.32421875, "completions/mean_terminated_length": 1249.1455078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2621934078633785, "epoch": 0.06713954033271562, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1332361251115799, "learning_rate": 1e-06, "loss": -0.0255, "num_tokens": 174330486.0, "reward": 0.57421875, "reward_std": 0.21093884110450745, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 394, "tools/generated_tokens": 4095.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1356.14453125, "completions/mean_terminated_length": 1148.944091796875, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.29124689288437366, "epoch": 0.06730994525741794, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4636945426464081, "learning_rate": 1e-06, "loss": 0.0261, "num_tokens": 174771403.0, "reward": 0.30859375, "reward_std": 0.29354849457740784, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 395, "tools/generated_tokens": 5324.171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1178.43359375, "completions/mean_terminated_length": 1120.4625244140625, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.2763081593438983, "epoch": 0.06748035018212026, "frac_reward_zero_std": 0.375, "grad_norm": 0.16088153421878815, "learning_rate": 1e-06, "loss": -0.0123, "num_tokens": 175163130.0, "reward": 0.6328125, "reward_std": 0.2552450895309448, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 396, "tools/generated_tokens": 4002.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1222.60546875, "completions/mean_terminated_length": 1160.1807861328125, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.28180971182882786, "epoch": 0.06765075510682259, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15223020315170288, "learning_rate": 1e-06, "loss": 0.0103, "num_tokens": 175548981.0, "reward": 0.38671875, "reward_std": 0.21840627491474152, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 397, "tools/generated_tokens": 3750.61328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1197.59375, "completions/mean_terminated_length": 1121.5999755859375, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.22976437583565712, "epoch": 0.06782116003152491, "frac_reward_zero_std": 0.5, "grad_norm": 0.16082558035850525, "learning_rate": 1e-06, "loss": -0.0028, "num_tokens": 175921229.0, "reward": 0.51953125, "reward_std": 0.17781277000904083, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 398, "tools/generated_tokens": 3109.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1363.28515625, "completions/mean_terminated_length": 1125.4368896484375, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.2488960139453411, "epoch": 0.06799156495622724, "frac_reward_zero_std": 0.375, "grad_norm": 0.16089093685150146, "learning_rate": 1e-06, "loss": 0.0171, "num_tokens": 176348262.0, "reward": 0.46875, "reward_std": 0.2441575825214386, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 399, "tools/generated_tokens": 4699.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1985.0, "completions/mean_length": 1397.7265625, "completions/mean_terminated_length": 1138.327880859375, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.277536628767848, "epoch": 0.06816196988092955, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1606374830007553, "learning_rate": 1e-06, "loss": 0.0506, "num_tokens": 176792672.0, "reward": 0.41015625, "reward_std": 0.29895496368408203, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 400, "tools/generated_tokens": 4837.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1170.4453125, "completions/mean_terminated_length": 1138.4696044921875, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.33787195198237896, "epoch": 0.06833237480563188, "frac_reward_zero_std": 0.25, "grad_norm": 0.18944306671619415, "learning_rate": 1e-06, "loss": 0.0022, "num_tokens": 177169746.0, "reward": 0.4921875, "reward_std": 0.3032139539718628, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 401, "tools/generated_tokens": 3754.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1225.8828125, "completions/mean_terminated_length": 1124.9210205078125, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.31717364117503166, "epoch": 0.06850277973033421, "frac_reward_zero_std": 0.5, "grad_norm": 0.18579499423503876, "learning_rate": 1e-06, "loss": 0.0167, "num_tokens": 177569028.0, "reward": 0.51953125, "reward_std": 0.1944383680820465, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 402, "tools/generated_tokens": 4225.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1217.55078125, "completions/mean_terminated_length": 1063.763916015625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2915894640609622, "epoch": 0.06867318465503654, "frac_reward_zero_std": 0.25, "grad_norm": 0.19436435401439667, "learning_rate": 1e-06, "loss": 0.0373, "num_tokens": 177950817.0, "reward": 0.6171875, "reward_std": 0.281505823135376, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 403, "tools/generated_tokens": 3849.55859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1212.453125, "completions/mean_terminated_length": 1048.471923828125, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "entropy": 0.2983880825340748, "epoch": 0.06884358957973885, "frac_reward_zero_std": 0.5, "grad_norm": 0.16463236510753632, "learning_rate": 1e-06, "loss": -0.0108, "num_tokens": 178344005.0, "reward": 0.453125, "reward_std": 0.21348227560520172, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 404, "tools/generated_tokens": 4068.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1428.3671875, "completions/mean_terminated_length": 1226.1036376953125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.3584884200245142, "epoch": 0.06901399450444118, "frac_reward_zero_std": 0.5, "grad_norm": 0.1689015030860901, "learning_rate": 1e-06, "loss": 0.0124, "num_tokens": 178790611.0, "reward": 0.3046875, "reward_std": 0.20624089241027832, "rewards/simpleverify_reward/mean": 0.3046875, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 405, "tools/generated_tokens": 4924.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.70703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1483.45703125, "completions/mean_terminated_length": 1283.3280029296875, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.26528845727443695, "epoch": 0.0691843994291435, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1359315812587738, "learning_rate": 1e-06, "loss": 0.0356, "num_tokens": 179252552.0, "reward": 0.30078125, "reward_std": 0.19966495037078857, "rewards/simpleverify_reward/mean": 0.30078125, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 406, "tools/generated_tokens": 5083.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1210.6484375, "completions/mean_terminated_length": 1127.991455078125, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.2559625366702676, "epoch": 0.06935480435384582, "frac_reward_zero_std": 0.375, "grad_norm": 0.15389494597911835, "learning_rate": 1e-06, "loss": 0.0007, "num_tokens": 179638510.0, "reward": 0.59375, "reward_std": 0.2542886435985565, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 407, "tools/generated_tokens": 3730.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1268.90234375, "completions/mean_terminated_length": 1124.625, "completions/min_length": 327.0, "completions/min_terminated_length": 327.0, "entropy": 0.2988923639059067, "epoch": 0.06952520927854815, "frac_reward_zero_std": 0.375, "grad_norm": 0.17689840495586395, "learning_rate": 1e-06, "loss": -0.0181, "num_tokens": 180049253.0, "reward": 0.47265625, "reward_std": 0.2333115190267563, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 408, "tools/generated_tokens": 4300.91796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1225.34765625, "completions/mean_terminated_length": 1112.0045166015625, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.2627219529822469, "epoch": 0.06969561420325047, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16810616850852966, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 180438814.0, "reward": 0.5625, "reward_std": 0.21519789099693298, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 409, "tools/generated_tokens": 3905.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1233.49609375, "completions/mean_terminated_length": 1137.462890625, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.2939452510327101, "epoch": 0.0698660191279528, "frac_reward_zero_std": 0.375, "grad_norm": 0.17348815500736237, "learning_rate": 1e-06, "loss": 0.0032, "num_tokens": 180828685.0, "reward": 0.54296875, "reward_std": 0.26400476694107056, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 410, "tools/generated_tokens": 3881.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1327.0078125, "completions/mean_terminated_length": 1181.4554443359375, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.23369611985981464, "epoch": 0.07003642405265512, "frac_reward_zero_std": 0.6875, "grad_norm": 0.11177127063274384, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 181241343.0, "reward": 0.37109375, "reward_std": 0.12806200981140137, "rewards/simpleverify_reward/mean": 0.37109375, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 411, "tools/generated_tokens": 4359.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1251.8046875, "completions/mean_terminated_length": 1108.709716796875, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.27532170712947845, "epoch": 0.07020682897735744, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2053053230047226, "learning_rate": 1e-06, "loss": -0.0021, "num_tokens": 181650989.0, "reward": 0.3671875, "reward_std": 0.26396670937538147, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 412, "tools/generated_tokens": 3947.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1393.99609375, "completions/mean_terminated_length": 1193.790771484375, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.30840983986854553, "epoch": 0.07037723390205977, "frac_reward_zero_std": 0.625, "grad_norm": 0.12739481031894684, "learning_rate": 1e-06, "loss": 0.0068, "num_tokens": 182089356.0, "reward": 0.42578125, "reward_std": 0.12709102034568787, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 413, "tools/generated_tokens": 4786.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1261.96484375, "completions/mean_terminated_length": 1149.6741943359375, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.28878416679799557, "epoch": 0.0705476388267621, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17790941894054413, "learning_rate": 1e-06, "loss": 0.0202, "num_tokens": 182495619.0, "reward": 0.59765625, "reward_std": 0.23958192765712738, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 414, "tools/generated_tokens": 4205.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1220.296875, "completions/mean_terminated_length": 1161.422607421875, "completions/min_length": 161.0, "completions/min_terminated_length": 161.0, "entropy": 0.25535366870462894, "epoch": 0.07071804375146441, "frac_reward_zero_std": 0.25, "grad_norm": 0.18710988759994507, "learning_rate": 1e-06, "loss": -0.0056, "num_tokens": 182890479.0, "reward": 0.56640625, "reward_std": 0.3078889548778534, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 415, "tools/generated_tokens": 4036.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1410.53125, "completions/mean_terminated_length": 1219.6141357421875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.2561458731070161, "epoch": 0.07088844867616674, "frac_reward_zero_std": 0.75, "grad_norm": 0.09066380560398102, "learning_rate": 1e-06, "loss": -0.0188, "num_tokens": 183313159.0, "reward": 0.31640625, "reward_std": 0.10341504216194153, "rewards/simpleverify_reward/mean": 0.31640625, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 416, "tools/generated_tokens": 3962.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1213.296875, "completions/mean_terminated_length": 1142.559326171875, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.2706059282645583, "epoch": 0.07105885360086907, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16542141139507294, "learning_rate": 1e-06, "loss": 0.0148, "num_tokens": 183698467.0, "reward": 0.484375, "reward_std": 0.2505345940589905, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 417, "tools/generated_tokens": 3821.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1220.109375, "completions/mean_terminated_length": 1080.2374267578125, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.2748273015022278, "epoch": 0.0712292585255714, "frac_reward_zero_std": 0.0625, "grad_norm": 0.2127537876367569, "learning_rate": 1e-06, "loss": 0.0378, "num_tokens": 184094063.0, "reward": 0.4296875, "reward_std": 0.3741224706172943, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 418, "tools/generated_tokens": 4364.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1334.8203125, "completions/mean_terminated_length": 1206.6451416015625, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.3021557554602623, "epoch": 0.07139966345027371, "frac_reward_zero_std": 0.25, "grad_norm": 0.17256174981594086, "learning_rate": 1e-06, "loss": 0.0096, "num_tokens": 184523761.0, "reward": 0.515625, "reward_std": 0.30045706033706665, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 419, "tools/generated_tokens": 4462.8203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1204.37890625, "completions/mean_terminated_length": 1100.7763671875, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.3174930810928345, "epoch": 0.07157006837497604, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1809980571269989, "learning_rate": 1e-06, "loss": -0.0065, "num_tokens": 184916194.0, "reward": 0.453125, "reward_std": 0.2715497612953186, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 420, "tools/generated_tokens": 4492.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1234.1796875, "completions/mean_terminated_length": 1187.09912109375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2462693229317665, "epoch": 0.07174047329967836, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13745740056037903, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 185301792.0, "reward": 0.5859375, "reward_std": 0.24063712358474731, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 421, "tools/generated_tokens": 3666.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1472.109375, "completions/mean_terminated_length": 1255.3763427734375, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.298484243452549, "epoch": 0.07191087822438068, "frac_reward_zero_std": 0.5, "grad_norm": 0.14522314071655273, "learning_rate": 1e-06, "loss": -0.0077, "num_tokens": 185750588.0, "reward": 0.36328125, "reward_std": 0.16516819596290588, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 422, "tools/generated_tokens": 4536.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1276.61328125, "completions/mean_terminated_length": 1185.663818359375, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.28468674700707197, "epoch": 0.072081283149083, "frac_reward_zero_std": 0.375, "grad_norm": 0.2937934994697571, "learning_rate": 1e-06, "loss": 0.021, "num_tokens": 186154649.0, "reward": 0.52734375, "reward_std": 0.25124263763427734, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 423, "tools/generated_tokens": 3772.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1099.46875, "completions/mean_terminated_length": 1076.7041015625, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.26107916329056025, "epoch": 0.07225168807378533, "frac_reward_zero_std": 0.25, "grad_norm": 0.19994470477104187, "learning_rate": 1e-06, "loss": 0.0205, "num_tokens": 186515265.0, "reward": 0.56640625, "reward_std": 0.32424497604370117, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 424, "tools/generated_tokens": 3523.4765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1287.51953125, "completions/mean_terminated_length": 1125.331787109375, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.25633655954152346, "epoch": 0.07242209299848766, "frac_reward_zero_std": 0.375, "grad_norm": 0.14257393777370453, "learning_rate": 1e-06, "loss": -0.0137, "num_tokens": 186934326.0, "reward": 0.53515625, "reward_std": 0.2603171467781067, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 425, "tools/generated_tokens": 4455.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1180.10546875, "completions/mean_terminated_length": 1098.508544921875, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.2560670170933008, "epoch": 0.07259249792318997, "frac_reward_zero_std": 0.375, "grad_norm": 0.19285213947296143, "learning_rate": 1e-06, "loss": -0.0027, "num_tokens": 187311873.0, "reward": 0.51953125, "reward_std": 0.20786382257938385, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 426, "tools/generated_tokens": 3868.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1213.4921875, "completions/mean_terminated_length": 1165.21484375, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.28291317261755466, "epoch": 0.0727629028478923, "frac_reward_zero_std": 0.375, "grad_norm": 0.15356355905532837, "learning_rate": 1e-06, "loss": 0.0003, "num_tokens": 187701711.0, "reward": 0.48046875, "reward_std": 0.2604663670063019, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 427, "tools/generated_tokens": 3693.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1270.1953125, "completions/mean_terminated_length": 1142.9227294921875, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.2575987661257386, "epoch": 0.07293330777259463, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17822514474391937, "learning_rate": 1e-06, "loss": 0.0224, "num_tokens": 188112161.0, "reward": 0.40234375, "reward_std": 0.30202803015708923, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 428, "tools/generated_tokens": 4302.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1259.6640625, "completions/mean_terminated_length": 1162.850830078125, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.27724962681531906, "epoch": 0.07310371269729696, "frac_reward_zero_std": 0.5, "grad_norm": 0.19279025495052338, "learning_rate": 1e-06, "loss": 0.0099, "num_tokens": 188509275.0, "reward": 0.4296875, "reward_std": 0.18181806802749634, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 429, "tools/generated_tokens": 3979.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1226.28125, "completions/mean_terminated_length": 1171.5042724609375, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.23214791808277369, "epoch": 0.07327411762199927, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14280451834201813, "learning_rate": 1e-06, "loss": 0.0115, "num_tokens": 188888307.0, "reward": 0.62109375, "reward_std": 0.2399258315563202, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 430, "tools/generated_tokens": 3314.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1379.44140625, "completions/mean_terminated_length": 1233.0, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.254175859503448, "epoch": 0.0734445225467016, "frac_reward_zero_std": 0.625, "grad_norm": 0.11756816506385803, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 189320612.0, "reward": 0.37890625, "reward_std": 0.14326362311840057, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 431, "tools/generated_tokens": 4315.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1216.16015625, "completions/mean_terminated_length": 1141.825439453125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24923141486942768, "epoch": 0.07361492747140393, "frac_reward_zero_std": 0.125, "grad_norm": 0.22574414312839508, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 189717629.0, "reward": 0.578125, "reward_std": 0.30700400471687317, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 432, "tools/generated_tokens": 4016.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1232.28515625, "completions/mean_terminated_length": 1094.47021484375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.29618942365050316, "epoch": 0.07378533239610625, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15020115673542023, "learning_rate": 1e-06, "loss": 0.0089, "num_tokens": 190117366.0, "reward": 0.48046875, "reward_std": 0.2005864679813385, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 433, "tools/generated_tokens": 4368.296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1123.3359375, "completions/mean_terminated_length": 1069.8470458984375, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.32763964496552944, "epoch": 0.07395573732080857, "frac_reward_zero_std": 0.375, "grad_norm": 0.17810696363449097, "learning_rate": 1e-06, "loss": -0.0048, "num_tokens": 190482876.0, "reward": 0.5, "reward_std": 0.2815170884132385, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 434, "tools/generated_tokens": 3739.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1209.07421875, "completions/mean_terminated_length": 1053.7176513671875, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.2867574654519558, "epoch": 0.0741261422455109, "frac_reward_zero_std": 0.5, "grad_norm": 0.15940769016742706, "learning_rate": 1e-06, "loss": 0.0198, "num_tokens": 190873311.0, "reward": 0.47265625, "reward_std": 0.19721892476081848, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 435, "tools/generated_tokens": 4113.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1183.37890625, "completions/mean_terminated_length": 1089.80517578125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.30609723739326, "epoch": 0.07429654717021322, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15700900554656982, "learning_rate": 1e-06, "loss": 0.0043, "num_tokens": 191261760.0, "reward": 0.5390625, "reward_std": 0.19366663694381714, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 436, "tools/generated_tokens": 3959.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1198.43359375, "completions/mean_terminated_length": 1152.9835205078125, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.2891850499436259, "epoch": 0.07446695209491554, "frac_reward_zero_std": 0.375, "grad_norm": 0.18941472470760345, "learning_rate": 1e-06, "loss": 0.0214, "num_tokens": 191650191.0, "reward": 0.5390625, "reward_std": 0.23095625638961792, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 437, "tools/generated_tokens": 4054.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1216.796875, "completions/mean_terminated_length": 1153.932861328125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2855268847197294, "epoch": 0.07463735701961786, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1926969736814499, "learning_rate": 1e-06, "loss": 0.009, "num_tokens": 192038699.0, "reward": 0.671875, "reward_std": 0.2702304720878601, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 438, "tools/generated_tokens": 3680.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1318.4296875, "completions/mean_terminated_length": 1171.1455078125, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.2969972314313054, "epoch": 0.07480776194432019, "frac_reward_zero_std": 0.25, "grad_norm": 0.17637549340724945, "learning_rate": 1e-06, "loss": 0.0505, "num_tokens": 192461145.0, "reward": 0.56640625, "reward_std": 0.30230605602264404, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 439, "tools/generated_tokens": 4486.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1308.47265625, "completions/mean_terminated_length": 1199.035888671875, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.2886236198246479, "epoch": 0.07497816686902252, "frac_reward_zero_std": 0.375, "grad_norm": 0.2587954103946686, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 192875586.0, "reward": 0.4296875, "reward_std": 0.24933947622776031, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 440, "tools/generated_tokens": 4124.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1242.453125, "completions/mean_terminated_length": 1195.8553466796875, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.25153734255582094, "epoch": 0.07514857179372483, "frac_reward_zero_std": 0.25, "grad_norm": 0.19918447732925415, "learning_rate": 1e-06, "loss": 0.0287, "num_tokens": 193265318.0, "reward": 0.56640625, "reward_std": 0.2962125539779663, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 441, "tools/generated_tokens": 3346.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.02734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1180.1328125, "completions/mean_terminated_length": 1028.8531494140625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25766815803945065, "epoch": 0.07531897671842716, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11928673088550568, "learning_rate": 1e-06, "loss": -0.0131, "num_tokens": 193647912.0, "reward": 0.4921875, "reward_std": 0.1331464648246765, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 442, "tools/generated_tokens": 3788.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1153.38671875, "completions/mean_terminated_length": 1081.6666259765625, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.26415857393294573, "epoch": 0.07548938164312949, "frac_reward_zero_std": 0.5, "grad_norm": 0.16005739569664001, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 194023099.0, "reward": 0.46875, "reward_std": 0.19588851928710938, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 443, "tools/generated_tokens": 3849.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1388.5625, "completions/mean_terminated_length": 1199.683349609375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.2570499451830983, "epoch": 0.07565978656783182, "frac_reward_zero_std": 0.1875, "grad_norm": 0.18124040961265564, "learning_rate": 1e-06, "loss": 0.0588, "num_tokens": 194464571.0, "reward": 0.46484375, "reward_std": 0.36255943775177, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 444, "tools/generated_tokens": 4996.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1201.9765625, "completions/mean_terminated_length": 1076.7802734375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.28643600922077894, "epoch": 0.07583019149253413, "frac_reward_zero_std": 0.4375, "grad_norm": 0.20645000040531158, "learning_rate": 1e-06, "loss": 0.0179, "num_tokens": 194853605.0, "reward": 0.421875, "reward_std": 0.21730193495750427, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 445, "tools/generated_tokens": 4297.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1238.63671875, "completions/mean_terminated_length": 1181.06689453125, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.2869391664862633, "epoch": 0.07600059641723646, "frac_reward_zero_std": 0.125, "grad_norm": 0.21641193330287933, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 195259576.0, "reward": 0.4609375, "reward_std": 0.35176295042037964, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 446, "tools/generated_tokens": 4310.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1249.00390625, "completions/mean_terminated_length": 1213.1346435546875, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.2864510640501976, "epoch": 0.07617100134193878, "frac_reward_zero_std": 0.25, "grad_norm": 0.1885346919298172, "learning_rate": 1e-06, "loss": 0.0213, "num_tokens": 195659401.0, "reward": 0.71875, "reward_std": 0.31807005405426025, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 447, "tools/generated_tokens": 3633.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1208.94921875, "completions/mean_terminated_length": 1130.064208984375, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.267677903175354, "epoch": 0.07634140626664111, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1536675989627838, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 196054684.0, "reward": 0.62890625, "reward_std": 0.24845923483371735, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 448, "tools/generated_tokens": 4080.9453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1090.375, "completions/mean_terminated_length": 1047.3795166015625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.28707336355000734, "epoch": 0.07651181119134343, "frac_reward_zero_std": 0.25, "grad_norm": 0.2251126617193222, "learning_rate": 1e-06, "loss": -0.0075, "num_tokens": 196415692.0, "reward": 0.66015625, "reward_std": 0.2708975076675415, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 449, "tools/generated_tokens": 3658.4140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1356.484375, "completions/mean_terminated_length": 1216.8826904296875, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.3043972812592983, "epoch": 0.07668221611604575, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17533689737319946, "learning_rate": 1e-06, "loss": 0.0191, "num_tokens": 196856664.0, "reward": 0.359375, "reward_std": 0.26049065589904785, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 450, "tools/generated_tokens": 4900.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1219.20703125, "completions/mean_terminated_length": 1087.954833984375, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.2809931878000498, "epoch": 0.07685262104074808, "frac_reward_zero_std": 0.5, "grad_norm": 0.16633576154708862, "learning_rate": 1e-06, "loss": 0.0253, "num_tokens": 197253501.0, "reward": 0.42578125, "reward_std": 0.17023906111717224, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 451, "tools/generated_tokens": 4251.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1251.37890625, "completions/mean_terminated_length": 1086.04248046875, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.24770265072584152, "epoch": 0.0770230259654504, "frac_reward_zero_std": 0.375, "grad_norm": 0.17578484117984772, "learning_rate": 1e-06, "loss": 0.0258, "num_tokens": 197647518.0, "reward": 0.609375, "reward_std": 0.2597846984863281, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 452, "tools/generated_tokens": 4115.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2011.0, "completions/mean_length": 1456.3125, "completions/mean_terminated_length": 1259.088623046875, "completions/min_length": 353.0, "completions/min_terminated_length": 353.0, "entropy": 0.2675662850961089, "epoch": 0.07719343089015272, "frac_reward_zero_std": 0.5, "grad_norm": 0.11961816996335983, "learning_rate": 1e-06, "loss": 0.0033, "num_tokens": 198102958.0, "reward": 0.46875, "reward_std": 0.17978152632713318, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 453, "tools/generated_tokens": 4840.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1183.78125, "completions/mean_terminated_length": 1118.420166015625, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.27085812017321587, "epoch": 0.07736383581485505, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1622786670923233, "learning_rate": 1e-06, "loss": 0.0143, "num_tokens": 198496246.0, "reward": 0.3828125, "reward_std": 0.29786184430122375, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 454, "tools/generated_tokens": 3959.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1303.62109375, "completions/mean_terminated_length": 1201.062255859375, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.2089649671688676, "epoch": 0.07753424073955738, "frac_reward_zero_std": 0.375, "grad_norm": 0.16132892668247223, "learning_rate": 1e-06, "loss": -0.0024, "num_tokens": 198909637.0, "reward": 0.4921875, "reward_std": 0.2361333966255188, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 455, "tools/generated_tokens": 4207.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1254.03125, "completions/mean_terminated_length": 1144.6400146484375, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.32212772220373154, "epoch": 0.07770464566425969, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19194775819778442, "learning_rate": 1e-06, "loss": 0.0269, "num_tokens": 199321965.0, "reward": 0.62890625, "reward_std": 0.3075515925884247, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 456, "tools/generated_tokens": 4414.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1311.28515625, "completions/mean_terminated_length": 1154.1658935546875, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.255185229703784, "epoch": 0.07787505058896202, "frac_reward_zero_std": 0.375, "grad_norm": 0.14567583799362183, "learning_rate": 1e-06, "loss": 0.0293, "num_tokens": 199748310.0, "reward": 0.4609375, "reward_std": 0.24079477787017822, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 457, "tools/generated_tokens": 4831.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1481.765625, "completions/mean_terminated_length": 1233.6517333984375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2644712319597602, "epoch": 0.07804545551366435, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16193120181560516, "learning_rate": 1e-06, "loss": 0.0378, "num_tokens": 200209066.0, "reward": 0.3828125, "reward_std": 0.246619313955307, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 458, "tools/generated_tokens": 5169.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1182.06640625, "completions/mean_terminated_length": 1071.4404296875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2632291382178664, "epoch": 0.07821586043836667, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15308211743831635, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 200590059.0, "reward": 0.57421875, "reward_std": 0.22797390818595886, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 459, "tools/generated_tokens": 3726.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1369.81640625, "completions/mean_terminated_length": 1188.5296630859375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.25467612966895103, "epoch": 0.07838626536306899, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14648422598838806, "learning_rate": 1e-06, "loss": 0.0222, "num_tokens": 201017660.0, "reward": 0.5, "reward_std": 0.22765710949897766, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 460, "tools/generated_tokens": 4697.83203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1451.5390625, "completions/mean_terminated_length": 1252.71875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.2917803544551134, "epoch": 0.07855667028777132, "frac_reward_zero_std": 0.75, "grad_norm": 0.09677249938249588, "learning_rate": 1e-06, "loss": 0.0202, "num_tokens": 201466006.0, "reward": 0.51171875, "reward_std": 0.10881631076335907, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 461, "tools/generated_tokens": 4659.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1271.828125, "completions/mean_terminated_length": 1160.946533203125, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.25832536444067955, "epoch": 0.07872707521247364, "frac_reward_zero_std": 0.6875, "grad_norm": 0.122776098549366, "learning_rate": 1e-06, "loss": 0.0221, "num_tokens": 201873818.0, "reward": 0.4140625, "reward_std": 0.15984314680099487, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 462, "tools/generated_tokens": 4095.8515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1353.0546875, "completions/mean_terminated_length": 1228.1658935546875, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.254767038859427, "epoch": 0.07889748013717597, "frac_reward_zero_std": 0.25, "grad_norm": 0.1598517894744873, "learning_rate": 1e-06, "loss": 0.015, "num_tokens": 202298888.0, "reward": 0.58203125, "reward_std": 0.28796231746673584, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 463, "tools/generated_tokens": 4529.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1132.15625, "completions/mean_terminated_length": 1083.160400390625, "completions/min_length": 319.0, "completions/min_terminated_length": 319.0, "entropy": 0.25874380860477686, "epoch": 0.07906788506187828, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16896232962608337, "learning_rate": 1e-06, "loss": 0.0008, "num_tokens": 202664832.0, "reward": 0.54296875, "reward_std": 0.2746518850326538, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 464, "tools/generated_tokens": 3596.16015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1342.31640625, "completions/mean_terminated_length": 1237.887939453125, "completions/min_length": 401.0, "completions/min_terminated_length": 401.0, "entropy": 0.29393049143254757, "epoch": 0.07923828998658061, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17599904537200928, "learning_rate": 1e-06, "loss": 0.021, "num_tokens": 203086417.0, "reward": 0.40625, "reward_std": 0.33642083406448364, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 465, "tools/generated_tokens": 4462.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1273.578125, "completions/mean_terminated_length": 1166.8800048828125, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.27936690114438534, "epoch": 0.07940869491128294, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17370912432670593, "learning_rate": 1e-06, "loss": 0.0063, "num_tokens": 203485589.0, "reward": 0.41796875, "reward_std": 0.22044281661510468, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 466, "tools/generated_tokens": 3977.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1350.75, "completions/mean_terminated_length": 1209.9906005859375, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.2872124407440424, "epoch": 0.07957909983598525, "frac_reward_zero_std": 0.25, "grad_norm": 0.19319450855255127, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 203908805.0, "reward": 0.53515625, "reward_std": 0.3216220736503601, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 467, "tools/generated_tokens": 4550.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1389.359375, "completions/mean_terminated_length": 1248.8909912109375, "completions/min_length": 421.0, "completions/min_terminated_length": 421.0, "entropy": 0.2329869018867612, "epoch": 0.07974950476068758, "frac_reward_zero_std": 0.25, "grad_norm": 0.17986977100372314, "learning_rate": 1e-06, "loss": 0.0259, "num_tokens": 204350673.0, "reward": 0.5703125, "reward_std": 0.31593742966651917, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 468, "tools/generated_tokens": 4605.3671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1333.98828125, "completions/mean_terminated_length": 1220.9140625, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.28531682677567005, "epoch": 0.07991990968538991, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18188199400901794, "learning_rate": 1e-06, "loss": 0.0017, "num_tokens": 204771134.0, "reward": 0.390625, "reward_std": 0.20597386360168457, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 469, "tools/generated_tokens": 4525.99609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1306.9140625, "completions/mean_terminated_length": 1197.251220703125, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.28891819529235363, "epoch": 0.08009031461009224, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13076718151569366, "learning_rate": 1e-06, "loss": -0.0272, "num_tokens": 205190696.0, "reward": 0.33203125, "reward_std": 0.16746041178703308, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 470, "tools/generated_tokens": 4394.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1403.59765625, "completions/mean_terminated_length": 1227.2686767578125, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.27110878843814135, "epoch": 0.08026071953479455, "frac_reward_zero_std": 0.375, "grad_norm": 0.15282106399536133, "learning_rate": 1e-06, "loss": 0.0136, "num_tokens": 205632289.0, "reward": 0.4453125, "reward_std": 0.2556079626083374, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 471, "tools/generated_tokens": 4979.60546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.74609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1357.50390625, "completions/mean_terminated_length": 1117.657958984375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.29090402089059353, "epoch": 0.08043112445949688, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16128665208816528, "learning_rate": 1e-06, "loss": 0.003, "num_tokens": 206064386.0, "reward": 0.296875, "reward_std": 0.2601088881492615, "rewards/simpleverify_reward/mean": 0.296875, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 472, "tools/generated_tokens": 4989.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1240.890625, "completions/mean_terminated_length": 1190.6556396484375, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "entropy": 0.26050027180463076, "epoch": 0.0806015293841992, "frac_reward_zero_std": 0.5, "grad_norm": 0.1341354250907898, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 206456918.0, "reward": 0.61328125, "reward_std": 0.20685215294361115, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 473, "tools/generated_tokens": 3656.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1108.60546875, "completions/mean_terminated_length": 1045.979248046875, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.28258791379630566, "epoch": 0.08077193430890153, "frac_reward_zero_std": 0.375, "grad_norm": 0.17450235784053802, "learning_rate": 1e-06, "loss": 0.0104, "num_tokens": 206825489.0, "reward": 0.42578125, "reward_std": 0.26895391941070557, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 474, "tools/generated_tokens": 4012.61328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1233.0859375, "completions/mean_terminated_length": 1213.528076171875, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.2500674147158861, "epoch": 0.08094233923360385, "frac_reward_zero_std": 0.5, "grad_norm": 0.16132110357284546, "learning_rate": 1e-06, "loss": -0.0027, "num_tokens": 207212983.0, "reward": 0.6953125, "reward_std": 0.19828036427497864, "rewards/simpleverify_reward/mean": 0.6953125, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 475, "tools/generated_tokens": 3041.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1251.67578125, "completions/mean_terminated_length": 1165.4935302734375, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.22036410216242075, "epoch": 0.08111274415830617, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13216455280780792, "learning_rate": 1e-06, "loss": 0.0018, "num_tokens": 207598996.0, "reward": 0.6328125, "reward_std": 0.14635254442691803, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 476, "tools/generated_tokens": 3235.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1377.046875, "completions/mean_terminated_length": 1222.2164306640625, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.29438223876059055, "epoch": 0.0812831490830085, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12111736834049225, "learning_rate": 1e-06, "loss": 0.0208, "num_tokens": 208025472.0, "reward": 0.5, "reward_std": 0.14139671623706818, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 477, "tools/generated_tokens": 4449.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1309.7265625, "completions/mean_terminated_length": 1185.0, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.2841954305768013, "epoch": 0.08145355400771083, "frac_reward_zero_std": 0.125, "grad_norm": 0.19221487641334534, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 208446010.0, "reward": 0.60546875, "reward_std": 0.29445403814315796, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 478, "tools/generated_tokens": 4565.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1241.0390625, "completions/mean_terminated_length": 1108.9908447265625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.28579610772430897, "epoch": 0.08162395893241314, "frac_reward_zero_std": 0.375, "grad_norm": 0.1440448760986328, "learning_rate": 1e-06, "loss": 0.0337, "num_tokens": 208845780.0, "reward": 0.5390625, "reward_std": 0.23425540328025818, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 479, "tools/generated_tokens": 4241.04296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1203.67578125, "completions/mean_terminated_length": 1091.5972900390625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.2577935494482517, "epoch": 0.08179436385711547, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14196252822875977, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 209231537.0, "reward": 0.42578125, "reward_std": 0.16461142897605896, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 480, "tools/generated_tokens": 3891.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1166.56640625, "completions/mean_terminated_length": 1107.80419921875, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.23823885526508093, "epoch": 0.0819647687818178, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15863071382045746, "learning_rate": 1e-06, "loss": 0.007, "num_tokens": 209620642.0, "reward": 0.5234375, "reward_std": 0.25263863801956177, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 481, "tools/generated_tokens": 3574.578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1248.40625, "completions/mean_terminated_length": 1134.1785888671875, "completions/min_length": 238.0, "completions/min_terminated_length": 238.0, "entropy": 0.29295533522963524, "epoch": 0.08213517370652011, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16509659588336945, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 210017354.0, "reward": 0.5, "reward_std": 0.20294174551963806, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 482, "tools/generated_tokens": 4112.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1316.91015625, "completions/mean_terminated_length": 1227.127197265625, "completions/min_length": 404.0, "completions/min_terminated_length": 404.0, "entropy": 0.23557352274656296, "epoch": 0.08230557863122244, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1472439020872116, "learning_rate": 1e-06, "loss": -0.0008, "num_tokens": 210426931.0, "reward": 0.52734375, "reward_std": 0.19322282075881958, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 483, "tools/generated_tokens": 3716.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1284.421875, "completions/mean_terminated_length": 1216.187255859375, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.2567291585728526, "epoch": 0.08247598355592477, "frac_reward_zero_std": 0.25, "grad_norm": 0.17313428223133087, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 210825343.0, "reward": 0.48046875, "reward_std": 0.2983798384666443, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 484, "tools/generated_tokens": 3572.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1321.9375, "completions/mean_terminated_length": 1175.3662109375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2825562469661236, "epoch": 0.0826463884806271, "frac_reward_zero_std": 0.375, "grad_norm": 0.16595816612243652, "learning_rate": 1e-06, "loss": 0.009, "num_tokens": 211246735.0, "reward": 0.49609375, "reward_std": 0.2654259204864502, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 485, "tools/generated_tokens": 4417.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1374.1171875, "completions/mean_terminated_length": 1284.677001953125, "completions/min_length": 189.0, "completions/min_terminated_length": 189.0, "entropy": 0.2517801756039262, "epoch": 0.08281679340532941, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14134158194065094, "learning_rate": 1e-06, "loss": 0.0198, "num_tokens": 211678525.0, "reward": 0.515625, "reward_std": 0.25053930282592773, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 486, "tools/generated_tokens": 4654.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1369.859375, "completions/mean_terminated_length": 1232.96240234375, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.24507506284862757, "epoch": 0.08298719833003174, "frac_reward_zero_std": 0.4375, "grad_norm": 0.12178443372249603, "learning_rate": 1e-06, "loss": -0.0033, "num_tokens": 212114009.0, "reward": 0.4765625, "reward_std": 0.2162114828824997, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 487, "tools/generated_tokens": 4409.87890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2007.0, "completions/mean_length": 1294.640625, "completions/mean_terminated_length": 1155.129638671875, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.30812946148216724, "epoch": 0.08315760325473406, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1886834055185318, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 212552525.0, "reward": 0.40234375, "reward_std": 0.32440072298049927, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 488, "tools/generated_tokens": 4702.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1274.91015625, "completions/mean_terminated_length": 1140.1513671875, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.26899847388267517, "epoch": 0.08332800817943639, "frac_reward_zero_std": 0.625, "grad_norm": 0.12434171885251999, "learning_rate": 1e-06, "loss": 0.006, "num_tokens": 212959094.0, "reward": 0.48046875, "reward_std": 0.12412451207637787, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 489, "tools/generated_tokens": 4018.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1238.53515625, "completions/mean_terminated_length": 1143.0960693359375, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.2552452450618148, "epoch": 0.0834984131041387, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17591121792793274, "learning_rate": 1e-06, "loss": 0.0046, "num_tokens": 213349135.0, "reward": 0.4375, "reward_std": 0.24153748154640198, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 490, "tools/generated_tokens": 3734.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1377.8515625, "completions/mean_terminated_length": 1190.2099609375, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.26415817346423864, "epoch": 0.08366881802884103, "frac_reward_zero_std": 0.25, "grad_norm": 0.31660452485084534, "learning_rate": 1e-06, "loss": 0.0221, "num_tokens": 213779177.0, "reward": 0.4609375, "reward_std": 0.2829289138317108, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 491, "tools/generated_tokens": 4633.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1330.4609375, "completions/mean_terminated_length": 1164.875, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.2483479054644704, "epoch": 0.08383922295354336, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16323445737361908, "learning_rate": 1e-06, "loss": 0.0353, "num_tokens": 214203935.0, "reward": 0.55078125, "reward_std": 0.3387996554374695, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 492, "tools/generated_tokens": 4642.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1271.5, "completions/mean_terminated_length": 1156.596435546875, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.2604938466101885, "epoch": 0.08400962787824569, "frac_reward_zero_std": 0.1875, "grad_norm": 0.2003244161605835, "learning_rate": 1e-06, "loss": 0.0277, "num_tokens": 214614687.0, "reward": 0.57421875, "reward_std": 0.30289244651794434, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 493, "tools/generated_tokens": 4375.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1401.3671875, "completions/mean_terminated_length": 1288.6558837890625, "completions/min_length": 412.0, "completions/min_terminated_length": 412.0, "entropy": 0.2736070640385151, "epoch": 0.084180032802948, "frac_reward_zero_std": 0.1875, "grad_norm": 0.18048396706581116, "learning_rate": 1e-06, "loss": 0.0188, "num_tokens": 215059581.0, "reward": 0.48046875, "reward_std": 0.3220744729042053, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 494, "tools/generated_tokens": 4505.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1332.9453125, "completions/mean_terminated_length": 1159.3931884765625, "completions/min_length": 348.0, "completions/min_terminated_length": 348.0, "entropy": 0.2770430566743016, "epoch": 0.08435043772765033, "frac_reward_zero_std": 0.375, "grad_norm": 0.15728560090065002, "learning_rate": 1e-06, "loss": 0.0319, "num_tokens": 215481007.0, "reward": 0.3984375, "reward_std": 0.2245136797428131, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 495, "tools/generated_tokens": 4612.9609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1418.55859375, "completions/mean_terminated_length": 1234.1767578125, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.247409513220191, "epoch": 0.08452084265235266, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11603175848722458, "learning_rate": 1e-06, "loss": 0.0355, "num_tokens": 215923390.0, "reward": 0.3828125, "reward_std": 0.1854248344898224, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 496, "tools/generated_tokens": 4498.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.50390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1181.125, "completions/mean_terminated_length": 1149.5384521484375, "completions/min_length": 191.0, "completions/min_terminated_length": 191.0, "entropy": 0.24172541499137878, "epoch": 0.08469124757705497, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1781323105096817, "learning_rate": 1e-06, "loss": -0.0071, "num_tokens": 216310734.0, "reward": 0.4296875, "reward_std": 0.2766585052013397, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 497, "tools/generated_tokens": 3885.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1421.66796875, "completions/mean_terminated_length": 1280.818115234375, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.25911517534404993, "epoch": 0.0848616525017573, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14875948429107666, "learning_rate": 1e-06, "loss": 0.0288, "num_tokens": 216754537.0, "reward": 0.484375, "reward_std": 0.2211625725030899, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 498, "tools/generated_tokens": 4517.66796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1181.16015625, "completions/mean_terminated_length": 1119.5062255859375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.2653562109917402, "epoch": 0.08503205742645963, "frac_reward_zero_std": 0.375, "grad_norm": 0.18100285530090332, "learning_rate": 1e-06, "loss": 0.0382, "num_tokens": 217134706.0, "reward": 0.640625, "reward_std": 0.23568323254585266, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 499, "tools/generated_tokens": 3621.17578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1359.6796875, "completions/mean_terminated_length": 1228.4232177734375, "completions/min_length": 332.0, "completions/min_terminated_length": 332.0, "entropy": 0.25509117916226387, "epoch": 0.08520246235116195, "frac_reward_zero_std": 0.5, "grad_norm": 0.15384499728679657, "learning_rate": 1e-06, "loss": 0.0148, "num_tokens": 217567216.0, "reward": 0.47265625, "reward_std": 0.20348459482192993, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 500, "tools/generated_tokens": 4735.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1266.765625, "completions/mean_terminated_length": 1170.8245849609375, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "entropy": 0.27620676439255476, "epoch": 0.08537286727586427, "frac_reward_zero_std": 0.1875, "grad_norm": 0.20635609328746796, "learning_rate": 1e-06, "loss": 0.0132, "num_tokens": 217969636.0, "reward": 0.59375, "reward_std": 0.33109644055366516, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 501, "tools/generated_tokens": 4042.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1342.703125, "completions/mean_terminated_length": 1262.973876953125, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.22437008377164602, "epoch": 0.0855432722005666, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15379106998443604, "learning_rate": 1e-06, "loss": 0.008, "num_tokens": 218394248.0, "reward": 0.59375, "reward_std": 0.2731332778930664, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 502, "tools/generated_tokens": 3902.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1311.43359375, "completions/mean_terminated_length": 1065.9114990234375, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "entropy": 0.29152974020689726, "epoch": 0.08571367712526892, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16427302360534668, "learning_rate": 1e-06, "loss": 0.0025, "num_tokens": 218821463.0, "reward": 0.390625, "reward_std": 0.21940404176712036, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 503, "tools/generated_tokens": 5111.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1223.25390625, "completions/mean_terminated_length": 1109.626708984375, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.30673097632825375, "epoch": 0.08588408204997125, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15234944224357605, "learning_rate": 1e-06, "loss": 0.0004, "num_tokens": 219224344.0, "reward": 0.3671875, "reward_std": 0.20818254351615906, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 504, "tools/generated_tokens": 4231.26171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1348.7265625, "completions/mean_terminated_length": 1139.2994384765625, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.29540538880974054, "epoch": 0.08605448697467356, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17678800225257874, "learning_rate": 1e-06, "loss": 0.0269, "num_tokens": 219661154.0, "reward": 0.46875, "reward_std": 0.18739622831344604, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 505, "tools/generated_tokens": 4868.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1341.7109375, "completions/mean_terminated_length": 1178.72119140625, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.29668791219592094, "epoch": 0.08622489189937589, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15114258229732513, "learning_rate": 1e-06, "loss": 0.0125, "num_tokens": 220086680.0, "reward": 0.484375, "reward_std": 0.1959541141986847, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 506, "tools/generated_tokens": 4781.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1269.5234375, "completions/mean_terminated_length": 1138.0045166015625, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.3113073166459799, "epoch": 0.08639529682407822, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13736459612846375, "learning_rate": 1e-06, "loss": 0.0316, "num_tokens": 220493294.0, "reward": 0.39453125, "reward_std": 0.17570874094963074, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 507, "tools/generated_tokens": 4709.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1381.125, "completions/mean_terminated_length": 1257.629638671875, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.27220352552831173, "epoch": 0.08656570174878055, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17900468409061432, "learning_rate": 1e-06, "loss": -0.001, "num_tokens": 220936398.0, "reward": 0.38671875, "reward_std": 0.3298344314098358, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 508, "tools/generated_tokens": 4741.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1280.109375, "completions/mean_terminated_length": 1107.42578125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2506442693993449, "epoch": 0.08673610667348286, "frac_reward_zero_std": 0.375, "grad_norm": 0.14439967274665833, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 221339642.0, "reward": 0.6015625, "reward_std": 0.2498009204864502, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 509, "tools/generated_tokens": 4032.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1274.8203125, "completions/mean_terminated_length": 1082.47314453125, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.24651102907955647, "epoch": 0.08690651159818519, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1554550975561142, "learning_rate": 1e-06, "loss": 0.0297, "num_tokens": 221744636.0, "reward": 0.32421875, "reward_std": 0.2420850694179535, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 510, "tools/generated_tokens": 4362.8203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1379.84765625, "completions/mean_terminated_length": 1175.3162841796875, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "entropy": 0.2568117417395115, "epoch": 0.08707691652288752, "frac_reward_zero_std": 0.375, "grad_norm": 0.18768467009067535, "learning_rate": 1e-06, "loss": 0.0504, "num_tokens": 222183605.0, "reward": 0.44140625, "reward_std": 0.2728801667690277, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 511, "tools/generated_tokens": 5155.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1418.23046875, "completions/mean_terminated_length": 1110.6685791015625, "completions/min_length": 257.0, "completions/min_terminated_length": 257.0, "entropy": 0.22845259215682745, "epoch": 0.08724732144758983, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1267181634902954, "learning_rate": 1e-06, "loss": 0.0095, "num_tokens": 222628096.0, "reward": 0.44921875, "reward_std": 0.18362826108932495, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 512, "tools/generated_tokens": 5154.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1265.6015625, "completions/mean_terminated_length": 1107.652587890625, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.2324536293745041, "epoch": 0.08741772637229216, "frac_reward_zero_std": 0.375, "grad_norm": 0.17009258270263672, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 223030538.0, "reward": 0.734375, "reward_std": 0.21204319596290588, "rewards/simpleverify_reward/mean": 0.734375, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 513, "tools/generated_tokens": 4585.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1289.421875, "completions/mean_terminated_length": 1157.2017822265625, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "entropy": 0.2680962225422263, "epoch": 0.08758813129699448, "frac_reward_zero_std": 0.25, "grad_norm": 0.17256851494312286, "learning_rate": 1e-06, "loss": -0.0061, "num_tokens": 223446518.0, "reward": 0.5390625, "reward_std": 0.31664419174194336, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 514, "tools/generated_tokens": 4801.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1303.2109375, "completions/mean_terminated_length": 1157.037353515625, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.29077679850161076, "epoch": 0.08775853622169681, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1605779081583023, "learning_rate": 1e-06, "loss": 0.031, "num_tokens": 223857244.0, "reward": 0.54296875, "reward_std": 0.23087677359580994, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 515, "tools/generated_tokens": 4199.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1261.7734375, "completions/mean_terminated_length": 1184.1630859375, "completions/min_length": 321.0, "completions/min_terminated_length": 321.0, "entropy": 0.26548791863024235, "epoch": 0.08792894114639913, "frac_reward_zero_std": 0.25, "grad_norm": 0.17304372787475586, "learning_rate": 1e-06, "loss": 0.0364, "num_tokens": 224258930.0, "reward": 0.625, "reward_std": 0.28111931681632996, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 516, "tools/generated_tokens": 4133.76953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1248.0859375, "completions/mean_terminated_length": 1091.0933837890625, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.30637590028345585, "epoch": 0.08809934607110145, "frac_reward_zero_std": 0.5, "grad_norm": 0.18615835905075073, "learning_rate": 1e-06, "loss": 0.0191, "num_tokens": 224662584.0, "reward": 0.5234375, "reward_std": 0.20882563292980194, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 517, "tools/generated_tokens": 4288.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1245.0, "completions/mean_terminated_length": 1208.950927734375, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.2723206877708435, "epoch": 0.08826975099580378, "frac_reward_zero_std": 0.5, "grad_norm": 0.13748545944690704, "learning_rate": 1e-06, "loss": -0.0147, "num_tokens": 225068904.0, "reward": 0.55078125, "reward_std": 0.19084002077579498, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 518, "tools/generated_tokens": 4053.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1372.375, "completions/mean_terminated_length": 1243.534912109375, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.26467883214354515, "epoch": 0.08844015592050611, "frac_reward_zero_std": 0.375, "grad_norm": 0.15883135795593262, "learning_rate": 1e-06, "loss": 0.0265, "num_tokens": 225498264.0, "reward": 0.62109375, "reward_std": 0.23502905666828156, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 519, "tools/generated_tokens": 4444.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1420.25390625, "completions/mean_terminated_length": 1244.4949951171875, "completions/min_length": 303.0, "completions/min_terminated_length": 303.0, "entropy": 0.22691770363599062, "epoch": 0.08861056084520842, "frac_reward_zero_std": 0.375, "grad_norm": 0.1373693346977234, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 225942569.0, "reward": 0.5625, "reward_std": 0.2284531146287918, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 520, "tools/generated_tokens": 4340.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.42578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1351.48046875, "completions/mean_terminated_length": 1173.936279296875, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.2516844943165779, "epoch": 0.08878096576991075, "frac_reward_zero_std": 0.25, "grad_norm": 0.159901425242424, "learning_rate": 1e-06, "loss": 0.0193, "num_tokens": 226375924.0, "reward": 0.50390625, "reward_std": 0.3148850202560425, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 521, "tools/generated_tokens": 4591.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1325.875, "completions/mean_terminated_length": 1167.6953125, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.277279500849545, "epoch": 0.08895137069461308, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15710198879241943, "learning_rate": 1e-06, "loss": 0.0055, "num_tokens": 226796996.0, "reward": 0.453125, "reward_std": 0.1650887131690979, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 522, "tools/generated_tokens": 4533.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1365.37890625, "completions/mean_terminated_length": 1242.69580078125, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.2397918114438653, "epoch": 0.0891217756193154, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15880653262138367, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 227216117.0, "reward": 0.515625, "reward_std": 0.24956358969211578, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 523, "tools/generated_tokens": 3845.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1225.85546875, "completions/mean_terminated_length": 1120.83251953125, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.20685587171465158, "epoch": 0.08929218054401772, "frac_reward_zero_std": 0.5, "grad_norm": 0.12243921309709549, "learning_rate": 1e-06, "loss": 0.0378, "num_tokens": 227601408.0, "reward": 0.6796875, "reward_std": 0.22662898898124695, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 524, "tools/generated_tokens": 3705.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1337.88671875, "completions/mean_terminated_length": 1217.92236328125, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.30382856726646423, "epoch": 0.08946258546872005, "frac_reward_zero_std": 0.5, "grad_norm": 0.14467647671699524, "learning_rate": 1e-06, "loss": -0.0024, "num_tokens": 228023443.0, "reward": 0.359375, "reward_std": 0.20192813873291016, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 525, "tools/generated_tokens": 4401.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1220.4609375, "completions/mean_terminated_length": 1089.4027099609375, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "entropy": 0.229482333175838, "epoch": 0.08963299039342237, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1602201908826828, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 228406825.0, "reward": 0.33984375, "reward_std": 0.18903234601020813, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 526, "tools/generated_tokens": 3828.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1407.58203125, "completions/mean_terminated_length": 1224.15576171875, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.2706261845305562, "epoch": 0.08980339531812469, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16631732881069183, "learning_rate": 1e-06, "loss": 0.009, "num_tokens": 228855758.0, "reward": 0.48828125, "reward_std": 0.2673723101615906, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 527, "tools/generated_tokens": 4951.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1294.7890625, "completions/mean_terminated_length": 1134.156494140625, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.22482021152973175, "epoch": 0.08997380024282702, "frac_reward_zero_std": 0.25, "grad_norm": 0.178030326962471, "learning_rate": 1e-06, "loss": 0.0273, "num_tokens": 229272232.0, "reward": 0.73828125, "reward_std": 0.29274123907089233, "rewards/simpleverify_reward/mean": 0.73828125, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 528, "tools/generated_tokens": 4430.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1346.6328125, "completions/mean_terminated_length": 1176.4078369140625, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.24884235206991434, "epoch": 0.09014420516752934, "frac_reward_zero_std": 0.5, "grad_norm": 0.12760142982006073, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 229693562.0, "reward": 0.62890625, "reward_std": 0.18332535028457642, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 529, "tools/generated_tokens": 4154.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1456.2890625, "completions/mean_terminated_length": 1246.529052734375, "completions/min_length": 332.0, "completions/min_terminated_length": 332.0, "entropy": 0.24968896806240082, "epoch": 0.09031461009223167, "frac_reward_zero_std": 0.1875, "grad_norm": 0.17811577022075653, "learning_rate": 1e-06, "loss": 0.0058, "num_tokens": 230152708.0, "reward": 0.34765625, "reward_std": 0.3287818431854248, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 530, "tools/generated_tokens": 5256.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1370.6953125, "completions/mean_terminated_length": 1256.2647705078125, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.26154812704771757, "epoch": 0.09048501501693398, "frac_reward_zero_std": 0.625, "grad_norm": 0.11092137545347214, "learning_rate": 1e-06, "loss": 0.0068, "num_tokens": 230576262.0, "reward": 0.59765625, "reward_std": 0.15151193737983704, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 531, "tools/generated_tokens": 4050.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1431.92578125, "completions/mean_terminated_length": 1263.3531494140625, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.26634883414953947, "epoch": 0.09065541994163631, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17712172865867615, "learning_rate": 1e-06, "loss": 0.0237, "num_tokens": 231030451.0, "reward": 0.56640625, "reward_std": 0.2361604869365692, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 532, "tools/generated_tokens": 4807.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1407.33203125, "completions/mean_terminated_length": 1202.587646484375, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.252150890417397, "epoch": 0.09082582486633864, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11143050342798233, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 231472984.0, "reward": 0.41796875, "reward_std": 0.17263562977313995, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 533, "tools/generated_tokens": 4871.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2007.0, "completions/mean_length": 1298.1875, "completions/mean_terminated_length": 1120.6956787109375, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.24661609530448914, "epoch": 0.09099622979104097, "frac_reward_zero_std": 0.25, "grad_norm": 0.18985028564929962, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 231877496.0, "reward": 0.6171875, "reward_std": 0.2607851028442383, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 534, "tools/generated_tokens": 4210.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1492.8359375, "completions/mean_terminated_length": 1249.561767578125, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.2084982069209218, "epoch": 0.09116663471574328, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13431037962436676, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 232331982.0, "reward": 0.55859375, "reward_std": 0.14171826839447021, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 535, "tools/generated_tokens": 4764.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2013.0, "completions/mean_length": 1331.87109375, "completions/mean_terminated_length": 1112.64794921875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.29038824141025543, "epoch": 0.09133703964044561, "frac_reward_zero_std": 0.25, "grad_norm": 0.16294419765472412, "learning_rate": 1e-06, "loss": 0.0589, "num_tokens": 232762781.0, "reward": 0.5234375, "reward_std": 0.28491154313087463, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 536, "tools/generated_tokens": 5019.875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1433.65234375, "completions/mean_terminated_length": 1277.058837890625, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 0.2674070904031396, "epoch": 0.09150744456514794, "frac_reward_zero_std": 0.625, "grad_norm": 0.10762283205986023, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 233212468.0, "reward": 0.54296875, "reward_std": 0.14274312555789948, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 537, "tools/generated_tokens": 4713.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2012.0, "completions/mean_length": 1267.76953125, "completions/mean_terminated_length": 1064.0738525390625, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.24824349116533995, "epoch": 0.09167784948985026, "frac_reward_zero_std": 0.1875, "grad_norm": 0.18577617406845093, "learning_rate": 1e-06, "loss": -0.0343, "num_tokens": 233626473.0, "reward": 0.58984375, "reward_std": 0.3347148895263672, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 538, "tools/generated_tokens": 4747.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1400.86328125, "completions/mean_terminated_length": 1243.791259765625, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.25223646126687527, "epoch": 0.09184825441455258, "frac_reward_zero_std": 0.375, "grad_norm": 0.14792688190937042, "learning_rate": 1e-06, "loss": 0.0154, "num_tokens": 234070438.0, "reward": 0.5078125, "reward_std": 0.21080546081066132, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 539, "tools/generated_tokens": 4536.875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1425.5703125, "completions/mean_terminated_length": 1182.016357421875, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.2843840243294835, "epoch": 0.0920186593392549, "frac_reward_zero_std": 0.5, "grad_norm": 0.1389995813369751, "learning_rate": 1e-06, "loss": 0.0402, "num_tokens": 234517000.0, "reward": 0.38671875, "reward_std": 0.2043357938528061, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 540, "tools/generated_tokens": 5017.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1396.09765625, "completions/mean_terminated_length": 1257.0711669921875, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.23499319050461054, "epoch": 0.09218906426395723, "frac_reward_zero_std": 0.5, "grad_norm": 0.13102610409259796, "learning_rate": 1e-06, "loss": 0.0186, "num_tokens": 234945249.0, "reward": 0.33984375, "reward_std": 0.20775945484638214, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 541, "tools/generated_tokens": 4012.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1371.35546875, "completions/mean_terminated_length": 1168.7156982421875, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.24373176600784063, "epoch": 0.09235946918865955, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16179470717906952, "learning_rate": 1e-06, "loss": 0.0246, "num_tokens": 235375692.0, "reward": 0.56640625, "reward_std": 0.18959103524684906, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 542, "tools/generated_tokens": 4483.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1401.87109375, "completions/mean_terminated_length": 1123.9329833984375, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.24623981583863497, "epoch": 0.09252987411336187, "frac_reward_zero_std": 0.25, "grad_norm": 0.17732244729995728, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 235816747.0, "reward": 0.46484375, "reward_std": 0.328233540058136, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 543, "tools/generated_tokens": 5241.875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1517.2890625, "completions/mean_terminated_length": 1347.7421875, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.255928092636168, "epoch": 0.0927002790380642, "frac_reward_zero_std": 0.375, "grad_norm": 0.13415896892547607, "learning_rate": 1e-06, "loss": 0.0056, "num_tokens": 236285221.0, "reward": 0.55859375, "reward_std": 0.229627788066864, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 544, "tools/generated_tokens": 4957.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1417.71484375, "completions/mean_terminated_length": 1279.6524658203125, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.28612892888486385, "epoch": 0.09287068396276653, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1610061079263687, "learning_rate": 1e-06, "loss": 0.0256, "num_tokens": 236731820.0, "reward": 0.42578125, "reward_std": 0.16926807165145874, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 545, "tools/generated_tokens": 4377.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1361.08984375, "completions/mean_terminated_length": 1214.59716796875, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.2613782323896885, "epoch": 0.09304108888746884, "frac_reward_zero_std": 0.125, "grad_norm": 0.18205983936786652, "learning_rate": 1e-06, "loss": 0.0454, "num_tokens": 237164915.0, "reward": 0.61328125, "reward_std": 0.34847772121429443, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 546, "tools/generated_tokens": 4737.1015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1422.04296875, "completions/mean_terminated_length": 1217.725341796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21857111807912588, "epoch": 0.09321149381217117, "frac_reward_zero_std": 0.75, "grad_norm": 0.08905293047428131, "learning_rate": 1e-06, "loss": 0.0179, "num_tokens": 237598270.0, "reward": 0.4375, "reward_std": 0.09011821448802948, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 547, "tools/generated_tokens": 4366.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1393.62890625, "completions/mean_terminated_length": 1206.2060546875, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.23382233548909426, "epoch": 0.0933818987368735, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13700401782989502, "learning_rate": 1e-06, "loss": 0.001, "num_tokens": 238030735.0, "reward": 0.390625, "reward_std": 0.20049379765987396, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 548, "tools/generated_tokens": 4417.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1373.2265625, "completions/mean_terminated_length": 1171.187744140625, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.3008579695597291, "epoch": 0.09355230366157583, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1843133270740509, "learning_rate": 1e-06, "loss": 0.0229, "num_tokens": 238468393.0, "reward": 0.58203125, "reward_std": 0.2604767680168152, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 549, "tools/generated_tokens": 4829.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1389.98828125, "completions/mean_terminated_length": 1127.51904296875, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.2668496873229742, "epoch": 0.09372270858627814, "frac_reward_zero_std": 0.25, "grad_norm": 0.1654992401599884, "learning_rate": 1e-06, "loss": -0.0029, "num_tokens": 238912774.0, "reward": 0.43359375, "reward_std": 0.3381253480911255, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 550, "tools/generated_tokens": 5118.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1324.28515625, "completions/mean_terminated_length": 1194.216552734375, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.25283054634928703, "epoch": 0.09389311351098047, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13537749648094177, "learning_rate": 1e-06, "loss": -0.0239, "num_tokens": 239338399.0, "reward": 0.37890625, "reward_std": 0.21953773498535156, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 551, "tools/generated_tokens": 4316.3125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1362.66796875, "completions/mean_terminated_length": 1166.371826171875, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.2578463824465871, "epoch": 0.0940635184356828, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1487479954957962, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 239762970.0, "reward": 0.54296875, "reward_std": 0.2552996277809143, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 552, "tools/generated_tokens": 4442.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.50390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1414.5390625, "completions/mean_terminated_length": 1268.3558349609375, "completions/min_length": 300.0, "completions/min_terminated_length": 300.0, "entropy": 0.24511006101965904, "epoch": 0.09423392336038512, "frac_reward_zero_std": 0.5, "grad_norm": 0.11943556368350983, "learning_rate": 1e-06, "loss": 0.0163, "num_tokens": 240209572.0, "reward": 0.4375, "reward_std": 0.18541164696216583, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 553, "tools/generated_tokens": 4622.55078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1462.5, "completions/mean_terminated_length": 1210.6424560546875, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.2894366355612874, "epoch": 0.09440432828508744, "frac_reward_zero_std": 0.375, "grad_norm": 0.15874801576137543, "learning_rate": 1e-06, "loss": -0.0072, "num_tokens": 240663108.0, "reward": 0.453125, "reward_std": 0.2290801852941513, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 554, "tools/generated_tokens": 4910.515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1432.4765625, "completions/mean_terminated_length": 1283.07763671875, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.2341524614021182, "epoch": 0.09457473320978976, "frac_reward_zero_std": 0.625, "grad_norm": 0.1132907047867775, "learning_rate": 1e-06, "loss": 0.0282, "num_tokens": 241101166.0, "reward": 0.640625, "reward_std": 0.1433543860912323, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 555, "tools/generated_tokens": 3984.484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1259.16015625, "completions/mean_terminated_length": 1062.9268798828125, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.25118235033005476, "epoch": 0.09474513813449209, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18437190353870392, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 241501047.0, "reward": 0.671875, "reward_std": 0.264829158782959, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 556, "tools/generated_tokens": 4419.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1393.91796875, "completions/mean_terminated_length": 1193.688720703125, "completions/min_length": 370.0, "completions/min_terminated_length": 370.0, "entropy": 0.27406329568475485, "epoch": 0.0949155430591944, "frac_reward_zero_std": 0.375, "grad_norm": 0.1724853217601776, "learning_rate": 1e-06, "loss": 0.0252, "num_tokens": 241932562.0, "reward": 0.51171875, "reward_std": 0.22209002077579498, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 557, "tools/generated_tokens": 4561.91796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1309.20703125, "completions/mean_terminated_length": 1151.6492919921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24596633110195398, "epoch": 0.09508594798389673, "frac_reward_zero_std": 0.5, "grad_norm": 0.16630980372428894, "learning_rate": 1e-06, "loss": -0.0091, "num_tokens": 242344311.0, "reward": 0.671875, "reward_std": 0.1959541141986847, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 558, "tools/generated_tokens": 3941.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1327.453125, "completions/mean_terminated_length": 1116.388916015625, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.2647271901369095, "epoch": 0.09525635290859906, "frac_reward_zero_std": 0.3125, "grad_norm": 0.26405781507492065, "learning_rate": 1e-06, "loss": 0.0272, "num_tokens": 242770811.0, "reward": 0.609375, "reward_std": 0.27398645877838135, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 559, "tools/generated_tokens": 4615.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1428.12890625, "completions/mean_terminated_length": 1250.6180419921875, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "entropy": 0.2787305386736989, "epoch": 0.09542675783330139, "frac_reward_zero_std": 0.375, "grad_norm": 0.15779337286949158, "learning_rate": 1e-06, "loss": 0.0039, "num_tokens": 243219868.0, "reward": 0.34375, "reward_std": 0.2779267430305481, "rewards/simpleverify_reward/mean": 0.34375, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 560, "tools/generated_tokens": 5172.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1295.26171875, "completions/mean_terminated_length": 1139.04248046875, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.23139166831970215, "epoch": 0.0955971627580037, "frac_reward_zero_std": 0.5, "grad_norm": 0.12333023548126221, "learning_rate": 1e-06, "loss": 0.0263, "num_tokens": 243627071.0, "reward": 0.7578125, "reward_std": 0.17978152632713318, "rewards/simpleverify_reward/mean": 0.7578125, "rewards/simpleverify_reward/std": 0.4292463958263397, "step": 561, "tools/generated_tokens": 3959.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1309.5, "completions/mean_terminated_length": 1164.57470703125, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.28400498628616333, "epoch": 0.09576756768270603, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18991196155548096, "learning_rate": 1e-06, "loss": 0.0227, "num_tokens": 244047359.0, "reward": 0.63671875, "reward_std": 0.272707998752594, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 562, "tools/generated_tokens": 4157.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1264.51953125, "completions/mean_terminated_length": 1140.4434814453125, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.27960452903062105, "epoch": 0.09593797260740836, "frac_reward_zero_std": 0.5, "grad_norm": 0.14789274334907532, "learning_rate": 1e-06, "loss": -0.0177, "num_tokens": 244446196.0, "reward": 0.546875, "reward_std": 0.19343584775924683, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 563, "tools/generated_tokens": 4008.53125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1345.1953125, "completions/mean_terminated_length": 1265.747802734375, "completions/min_length": 361.0, "completions/min_terminated_length": 361.0, "entropy": 0.29170692525804043, "epoch": 0.09610837753211068, "frac_reward_zero_std": 0.625, "grad_norm": 0.1148810014128685, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 244868278.0, "reward": 0.42578125, "reward_std": 0.14161168038845062, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 564, "tools/generated_tokens": 4537.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1373.40625, "completions/mean_terminated_length": 1283.86279296875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2389817675575614, "epoch": 0.096278782456813, "frac_reward_zero_std": 0.375, "grad_norm": 0.1451931744813919, "learning_rate": 1e-06, "loss": -0.001, "num_tokens": 245283166.0, "reward": 0.68359375, "reward_std": 0.25705814361572266, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 565, "tools/generated_tokens": 3573.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.07421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1485.703125, "completions/mean_terminated_length": 1191.1905517578125, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.2955512637272477, "epoch": 0.09644918738151533, "frac_reward_zero_std": 0.5, "grad_norm": 0.1512872725725174, "learning_rate": 1e-06, "loss": 0.0033, "num_tokens": 245744994.0, "reward": 0.33203125, "reward_std": 0.1676161289215088, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 566, "tools/generated_tokens": 5485.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1331.4609375, "completions/mean_terminated_length": 1202.6866455078125, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.26168250665068626, "epoch": 0.09661959230621765, "frac_reward_zero_std": 0.25, "grad_norm": 0.16792535781860352, "learning_rate": 1e-06, "loss": 0.0394, "num_tokens": 246161976.0, "reward": 0.66015625, "reward_std": 0.27168312668800354, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 567, "tools/generated_tokens": 4283.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1477.09375, "completions/mean_terminated_length": 1286.8021240234375, "completions/min_length": 408.0, "completions/min_terminated_length": 408.0, "entropy": 0.2853711638599634, "epoch": 0.09678999723091998, "frac_reward_zero_std": 0.25, "grad_norm": 0.17652322351932526, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 246615088.0, "reward": 0.51171875, "reward_std": 0.3131811320781708, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 568, "tools/generated_tokens": 4989.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1488.10546875, "completions/mean_terminated_length": 1194.83935546875, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.26467016711831093, "epoch": 0.0969604021556223, "frac_reward_zero_std": 0.5, "grad_norm": 0.1525377780199051, "learning_rate": 1e-06, "loss": 0.005, "num_tokens": 247072363.0, "reward": 0.49609375, "reward_std": 0.20377904176712036, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 569, "tools/generated_tokens": 4888.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1282.50390625, "completions/mean_terminated_length": 1140.75, "completions/min_length": 298.0, "completions/min_terminated_length": 298.0, "entropy": 0.2252415968105197, "epoch": 0.09713080708032462, "frac_reward_zero_std": 0.375, "grad_norm": 0.1339685469865799, "learning_rate": 1e-06, "loss": -0.0123, "num_tokens": 247492268.0, "reward": 0.28515625, "reward_std": 0.2525572180747986, "rewards/simpleverify_reward/mean": 0.28515625, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 570, "tools/generated_tokens": 4410.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2006.0, "completions/mean_length": 1325.890625, "completions/mean_terminated_length": 1240.7510986328125, "completions/min_length": 191.0, "completions/min_terminated_length": 191.0, "entropy": 0.25900744181126356, "epoch": 0.09730121200502695, "frac_reward_zero_std": 0.25, "grad_norm": 0.18688349425792694, "learning_rate": 1e-06, "loss": 0.0256, "num_tokens": 247899936.0, "reward": 0.578125, "reward_std": 0.311518132686615, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 571, "tools/generated_tokens": 4149.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1395.62890625, "completions/mean_terminated_length": 1204.5302734375, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.3108964003622532, "epoch": 0.09747161692972926, "frac_reward_zero_std": 0.375, "grad_norm": 0.3480401039123535, "learning_rate": 1e-06, "loss": -0.0001, "num_tokens": 248347121.0, "reward": 0.53515625, "reward_std": 0.2348030060529709, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 572, "tools/generated_tokens": 4563.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1355.421875, "completions/mean_terminated_length": 1143.4080810546875, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.3062203638255596, "epoch": 0.09764202185443159, "frac_reward_zero_std": 0.375, "grad_norm": 0.13149289786815643, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 248771501.0, "reward": 0.55078125, "reward_std": 0.22996041178703308, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 573, "tools/generated_tokens": 4691.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1373.05078125, "completions/mean_terminated_length": 1152.7305908203125, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.2653664303943515, "epoch": 0.09781242677913392, "frac_reward_zero_std": 0.125, "grad_norm": 0.1707451492547989, "learning_rate": 1e-06, "loss": 0.0444, "num_tokens": 249209626.0, "reward": 0.61328125, "reward_std": 0.3082984387874603, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 574, "tools/generated_tokens": 5197.05078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.35546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1486.21484375, "completions/mean_terminated_length": 1176.39990234375, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.25862658116966486, "epoch": 0.09798283170383625, "frac_reward_zero_std": 0.375, "grad_norm": 0.15441524982452393, "learning_rate": 1e-06, "loss": 0.0235, "num_tokens": 249675457.0, "reward": 0.4609375, "reward_std": 0.25491246581077576, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 575, "tools/generated_tokens": 5558.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.98828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1417.859375, "completions/mean_terminated_length": 1237.3668212890625, "completions/min_length": 416.0, "completions/min_terminated_length": 416.0, "entropy": 0.26523235253989697, "epoch": 0.09815323662853856, "frac_reward_zero_std": 0.25, "grad_norm": 0.17353393137454987, "learning_rate": 1e-06, "loss": 0.0453, "num_tokens": 250119933.0, "reward": 0.47265625, "reward_std": 0.2930987477302551, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 576, "tools/generated_tokens": 5121.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1413.68359375, "completions/mean_terminated_length": 1140.8323974609375, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.29382974095642567, "epoch": 0.09832364155324089, "frac_reward_zero_std": 0.5, "grad_norm": 0.1510627418756485, "learning_rate": 1e-06, "loss": 0.0405, "num_tokens": 250565788.0, "reward": 0.39453125, "reward_std": 0.19226783514022827, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 577, "tools/generated_tokens": 5093.7421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1327.44921875, "completions/mean_terminated_length": 1121.0703125, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.24083214346319437, "epoch": 0.09849404647794321, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16204893589019775, "learning_rate": 1e-06, "loss": 0.0186, "num_tokens": 250983343.0, "reward": 0.64453125, "reward_std": 0.25978732109069824, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 578, "tools/generated_tokens": 4263.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1393.5390625, "completions/mean_terminated_length": 1170.8272705078125, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "entropy": 0.2650506068021059, "epoch": 0.09866445140264554, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1399562507867813, "learning_rate": 1e-06, "loss": 0.0208, "num_tokens": 251413289.0, "reward": 0.36328125, "reward_std": 0.2348029911518097, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 579, "tools/generated_tokens": 4833.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1326.2890625, "completions/mean_terminated_length": 1172.37451171875, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.2585566472262144, "epoch": 0.09883485632734786, "frac_reward_zero_std": 0.25, "grad_norm": 0.16252174973487854, "learning_rate": 1e-06, "loss": 0.0422, "num_tokens": 251837987.0, "reward": 0.70703125, "reward_std": 0.2803495228290558, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 580, "tools/generated_tokens": 4486.29296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1351.0078125, "completions/mean_terminated_length": 1118.682373046875, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.27031111624091864, "epoch": 0.09900526125205018, "frac_reward_zero_std": 0.25, "grad_norm": 0.1783943921327591, "learning_rate": 1e-06, "loss": 0.0044, "num_tokens": 252273317.0, "reward": 0.51953125, "reward_std": 0.31449854373931885, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 581, "tools/generated_tokens": 5079.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1314.05859375, "completions/mean_terminated_length": 1064.2984619140625, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.27539417054504156, "epoch": 0.09917566617675251, "frac_reward_zero_std": 0.5, "grad_norm": 0.14385437965393066, "learning_rate": 1e-06, "loss": 0.0234, "num_tokens": 252692804.0, "reward": 0.421875, "reward_std": 0.18023642897605896, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 582, "tools/generated_tokens": 4722.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1354.72265625, "completions/mean_terminated_length": 1142.5, "completions/min_length": 211.0, "completions/min_terminated_length": 211.0, "entropy": 0.2697906754910946, "epoch": 0.09934607110145484, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11965537816286087, "learning_rate": 1e-06, "loss": 0.0106, "num_tokens": 253122493.0, "reward": 0.49609375, "reward_std": 0.19038984179496765, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 583, "tools/generated_tokens": 5066.7421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1370.55078125, "completions/mean_terminated_length": 1154.04638671875, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.24087819084525108, "epoch": 0.09951647602615715, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13983558118343353, "learning_rate": 1e-06, "loss": 0.0118, "num_tokens": 253552842.0, "reward": 0.63671875, "reward_std": 0.17177122831344604, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 584, "tools/generated_tokens": 4498.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1396.62109375, "completions/mean_terminated_length": 1242.429931640625, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.2532341908663511, "epoch": 0.09968688095085948, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1424928903579712, "learning_rate": 1e-06, "loss": 0.028, "num_tokens": 253988489.0, "reward": 0.578125, "reward_std": 0.20938239991664886, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 585, "tools/generated_tokens": 4484.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1379.99609375, "completions/mean_terminated_length": 1157.3333740234375, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.2904137782752514, "epoch": 0.09985728587556181, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1823052614927292, "learning_rate": 1e-06, "loss": 0.0017, "num_tokens": 254409576.0, "reward": 0.62109375, "reward_std": 0.2342825084924698, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 586, "tools/generated_tokens": 4236.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1428.69140625, "completions/mean_terminated_length": 1181.644775390625, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.2986216712743044, "epoch": 0.10002769080026412, "frac_reward_zero_std": 0.5, "grad_norm": 0.12949174642562866, "learning_rate": 1e-06, "loss": 0.0378, "num_tokens": 254860249.0, "reward": 0.49609375, "reward_std": 0.18408125638961792, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 587, "tools/generated_tokens": 5332.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.90625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1283.09765625, "completions/mean_terminated_length": 1192.9127197265625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2757903980091214, "epoch": 0.10019809572496645, "frac_reward_zero_std": 0.375, "grad_norm": 0.14384324848651886, "learning_rate": 1e-06, "loss": 0.005, "num_tokens": 255253570.0, "reward": 0.57421875, "reward_std": 0.22808241844177246, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 588, "tools/generated_tokens": 3539.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1481.046875, "completions/mean_terminated_length": 1204.18017578125, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.28604976274073124, "epoch": 0.10036850064966878, "frac_reward_zero_std": 0.1875, "grad_norm": 0.23569124937057495, "learning_rate": 1e-06, "loss": 0.043, "num_tokens": 255712222.0, "reward": 0.546875, "reward_std": 0.29567813873291016, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 589, "tools/generated_tokens": 5297.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.86328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2000.0, "completions/mean_length": 1219.1171875, "completions/mean_terminated_length": 1012.9121704101562, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.28153133019804955, "epoch": 0.1005389055743711, "frac_reward_zero_std": 0.375, "grad_norm": 0.16634711623191833, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 256111612.0, "reward": 0.5625, "reward_std": 0.2226376235485077, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 590, "tools/generated_tokens": 4539.125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1331.47265625, "completions/mean_terminated_length": 1072.3138427734375, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.2969023184850812, "epoch": 0.10070931049907342, "frac_reward_zero_std": 0.375, "grad_norm": 0.15822678804397583, "learning_rate": 1e-06, "loss": 0.03, "num_tokens": 256538581.0, "reward": 0.40625, "reward_std": 0.2191779911518097, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 591, "tools/generated_tokens": 4739.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1235.59375, "completions/mean_terminated_length": 1106.94580078125, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.2496339399367571, "epoch": 0.10087971542377575, "frac_reward_zero_std": 0.375, "grad_norm": 0.13325046002864838, "learning_rate": 1e-06, "loss": 0.0132, "num_tokens": 256934813.0, "reward": 0.4140625, "reward_std": 0.2615154981613159, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 592, "tools/generated_tokens": 4283.61328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1467.3203125, "completions/mean_terminated_length": 1222.14453125, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.2549753934144974, "epoch": 0.10105012034847807, "frac_reward_zero_std": 0.3125, "grad_norm": 0.13179758191108704, "learning_rate": 1e-06, "loss": -0.0164, "num_tokens": 257393599.0, "reward": 0.5625, "reward_std": 0.25197336077690125, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 593, "tools/generated_tokens": 4995.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1316.12890625, "completions/mean_terminated_length": 1168.3802490234375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.26244362629950047, "epoch": 0.1012205252731804, "frac_reward_zero_std": 0.125, "grad_norm": 0.19303129613399506, "learning_rate": 1e-06, "loss": 0.0399, "num_tokens": 257816256.0, "reward": 0.625, "reward_std": 0.3593369722366333, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 594, "tools/generated_tokens": 4332.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1395.97265625, "completions/mean_terminated_length": 1221.6683349609375, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.2524370811879635, "epoch": 0.10139093019788271, "frac_reward_zero_std": 0.375, "grad_norm": 0.15308091044425964, "learning_rate": 1e-06, "loss": 0.027, "num_tokens": 258260297.0, "reward": 0.515625, "reward_std": 0.24872365593910217, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 595, "tools/generated_tokens": 5123.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1274.6171875, "completions/mean_terminated_length": 1077.4853515625, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.2460261918604374, "epoch": 0.10156133512258504, "frac_reward_zero_std": 0.625, "grad_norm": 0.11448148638010025, "learning_rate": 1e-06, "loss": 0.0518, "num_tokens": 258672919.0, "reward": 0.4453125, "reward_std": 0.1364503651857376, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 596, "tools/generated_tokens": 4514.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1299.58984375, "completions/mean_terminated_length": 1094.801025390625, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "entropy": 0.28133365977555513, "epoch": 0.10173174004728737, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12616626918315887, "learning_rate": 1e-06, "loss": 0.0215, "num_tokens": 259080366.0, "reward": 0.484375, "reward_std": 0.13896197080612183, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 597, "tools/generated_tokens": 4363.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1217.3671875, "completions/mean_terminated_length": 1158.2845458984375, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.24622021056711674, "epoch": 0.1019021449719897, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16811169683933258, "learning_rate": 1e-06, "loss": 0.0237, "num_tokens": 259475324.0, "reward": 0.69921875, "reward_std": 0.2175418734550476, "rewards/simpleverify_reward/mean": 0.69921875, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 598, "tools/generated_tokens": 3745.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1126.17578125, "completions/mean_terminated_length": 994.4866333007812, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.25988560542464256, "epoch": 0.10207254989669201, "frac_reward_zero_std": 0.375, "grad_norm": 0.21057769656181335, "learning_rate": 1e-06, "loss": 0.0157, "num_tokens": 259847241.0, "reward": 0.6015625, "reward_std": 0.26239442825317383, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 599, "tools/generated_tokens": 3814.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1431.1484375, "completions/mean_terminated_length": 1199.00537109375, "completions/min_length": 280.0, "completions/min_terminated_length": 280.0, "entropy": 0.23735546227544546, "epoch": 0.10224295482139434, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15083478391170502, "learning_rate": 1e-06, "loss": 0.0431, "num_tokens": 260298815.0, "reward": 0.515625, "reward_std": 0.31496453285217285, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 600, "tools/generated_tokens": 5311.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.89453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1336.73828125, "completions/mean_terminated_length": 1119.0101318359375, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.28465794399380684, "epoch": 0.10241335974609667, "frac_reward_zero_std": 0.375, "grad_norm": 0.1729150116443634, "learning_rate": 1e-06, "loss": 0.0123, "num_tokens": 260723436.0, "reward": 0.51953125, "reward_std": 0.25559213757514954, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 601, "tools/generated_tokens": 4544.7578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1504.16015625, "completions/mean_terminated_length": 1238.5814208984375, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.2872716346755624, "epoch": 0.10258376467079898, "frac_reward_zero_std": 0.5, "grad_norm": 0.13313445448875427, "learning_rate": 1e-06, "loss": 0.0354, "num_tokens": 261199509.0, "reward": 0.2421875, "reward_std": 0.20465734601020813, "rewards/simpleverify_reward/mean": 0.2421875, "rewards/simpleverify_reward/std": 0.4292463958263397, "step": 602, "tools/generated_tokens": 5704.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.05078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1308.875, "completions/mean_terminated_length": 1106.6268310546875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.3057608436793089, "epoch": 0.10275416959550131, "frac_reward_zero_std": 0.375, "grad_norm": 0.1824343204498291, "learning_rate": 1e-06, "loss": 0.0293, "num_tokens": 261613621.0, "reward": 0.578125, "reward_std": 0.2079564929008484, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 603, "tools/generated_tokens": 4460.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1194.90234375, "completions/mean_terminated_length": 1110.6995849609375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.2693713651970029, "epoch": 0.10292457452020363, "frac_reward_zero_std": 0.375, "grad_norm": 0.17988616228103638, "learning_rate": 1e-06, "loss": 0.0239, "num_tokens": 261994300.0, "reward": 0.578125, "reward_std": 0.2487104833126068, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 604, "tools/generated_tokens": 3602.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1287.71875, "completions/mean_terminated_length": 1116.7607421875, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "entropy": 0.27751616202294827, "epoch": 0.10309497944490596, "frac_reward_zero_std": 0.5, "grad_norm": 0.14557887613773346, "learning_rate": 1e-06, "loss": 0.0268, "num_tokens": 262404740.0, "reward": 0.59375, "reward_std": 0.1849614679813385, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 605, "tools/generated_tokens": 4519.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1460.26171875, "completions/mean_terminated_length": 1225.819580078125, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "entropy": 0.25700395181775093, "epoch": 0.10326538436960828, "frac_reward_zero_std": 0.4375, "grad_norm": 0.135623499751091, "learning_rate": 1e-06, "loss": 0.017, "num_tokens": 262861479.0, "reward": 0.41015625, "reward_std": 0.2205064743757248, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 606, "tools/generated_tokens": 4972.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1311.9765625, "completions/mean_terminated_length": 1101.1708984375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2471226779744029, "epoch": 0.1034357892943106, "frac_reward_zero_std": 0.625, "grad_norm": 0.12482727319002151, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 263274897.0, "reward": 0.47265625, "reward_std": 0.14502215385437012, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 607, "tools/generated_tokens": 4319.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1299.6796875, "completions/mean_terminated_length": 1075.5634765625, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.26958257611840963, "epoch": 0.10360619421901293, "frac_reward_zero_std": 0.375, "grad_norm": 0.17142032086849213, "learning_rate": 1e-06, "loss": 0.0147, "num_tokens": 263694703.0, "reward": 0.421875, "reward_std": 0.24933947622776031, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 608, "tools/generated_tokens": 4915.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1339.625, "completions/mean_terminated_length": 1127.4771728515625, "completions/min_length": 326.0, "completions/min_terminated_length": 326.0, "entropy": 0.3016281109303236, "epoch": 0.10377659914371526, "frac_reward_zero_std": 0.125, "grad_norm": 0.2015984207391739, "learning_rate": 1e-06, "loss": 0.0301, "num_tokens": 264127599.0, "reward": 0.51171875, "reward_std": 0.36788105964660645, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 609, "tools/generated_tokens": 4923.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.39453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1563.81640625, "completions/mean_terminated_length": 1248.3289794921875, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.309479346498847, "epoch": 0.10394700406841757, "frac_reward_zero_std": 0.625, "grad_norm": 0.138493612408638, "learning_rate": 1e-06, "loss": 0.0069, "num_tokens": 264604912.0, "reward": 0.328125, "reward_std": 0.15284234285354614, "rewards/simpleverify_reward/mean": 0.328125, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 610, "tools/generated_tokens": 5363.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1327.5859375, "completions/mean_terminated_length": 1152.7379150390625, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.2610814590007067, "epoch": 0.1041174089931199, "frac_reward_zero_std": 0.375, "grad_norm": 0.15999998152256012, "learning_rate": 1e-06, "loss": 0.034, "num_tokens": 265025798.0, "reward": 0.453125, "reward_std": 0.23755928874015808, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 611, "tools/generated_tokens": 4703.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1418.15625, "completions/mean_terminated_length": 1137.0509033203125, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.2829169724136591, "epoch": 0.10428781391782223, "frac_reward_zero_std": 0.25, "grad_norm": 0.17679709196090698, "learning_rate": 1e-06, "loss": 0.0353, "num_tokens": 265469566.0, "reward": 0.375, "reward_std": 0.2794036865234375, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 612, "tools/generated_tokens": 5362.171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1340.7109375, "completions/mean_terminated_length": 1173.294677734375, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.26936994958668947, "epoch": 0.10445821884252456, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1563178300857544, "learning_rate": 1e-06, "loss": 0.0342, "num_tokens": 265894980.0, "reward": 0.57421875, "reward_std": 0.21658216416835785, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 613, "tools/generated_tokens": 4540.72265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2005.0, "completions/mean_length": 1345.85546875, "completions/mean_terminated_length": 1238.31982421875, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "entropy": 0.2717377059161663, "epoch": 0.10462862376722687, "frac_reward_zero_std": 0.375, "grad_norm": 0.18921178579330444, "learning_rate": 1e-06, "loss": 0.0413, "num_tokens": 266306623.0, "reward": 0.71875, "reward_std": 0.2729611396789551, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 614, "tools/generated_tokens": 4025.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1265.1484375, "completions/mean_terminated_length": 1079.840576171875, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.2637931974604726, "epoch": 0.1047990286919292, "frac_reward_zero_std": 0.375, "grad_norm": 0.1937050074338913, "learning_rate": 1e-06, "loss": 0.026, "num_tokens": 266707573.0, "reward": 0.54296875, "reward_std": 0.22175738215446472, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 615, "tools/generated_tokens": 4193.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1381.265625, "completions/mean_terminated_length": 1089.10107421875, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.31916841957718134, "epoch": 0.10496943361663152, "frac_reward_zero_std": 0.375, "grad_norm": 0.1596374660730362, "learning_rate": 1e-06, "loss": 0.0115, "num_tokens": 267149529.0, "reward": 0.26171875, "reward_std": 0.2327008694410324, "rewards/simpleverify_reward/mean": 0.26171875, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 616, "tools/generated_tokens": 5253.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1441.234375, "completions/mean_terminated_length": 1279.0445556640625, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.28609442338347435, "epoch": 0.10513983854133384, "frac_reward_zero_std": 0.4375, "grad_norm": 0.22561423480510712, "learning_rate": 1e-06, "loss": 0.0367, "num_tokens": 267589349.0, "reward": 0.6796875, "reward_std": 0.18386822938919067, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 617, "tools/generated_tokens": 4337.2578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1442.87109375, "completions/mean_terminated_length": 1236.9423828125, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.2818992603570223, "epoch": 0.10531024346603617, "frac_reward_zero_std": 0.5, "grad_norm": 0.12229091674089432, "learning_rate": 1e-06, "loss": -0.0134, "num_tokens": 268048260.0, "reward": 0.46875, "reward_std": 0.19918768107891083, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 618, "tools/generated_tokens": 5098.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1303.8515625, "completions/mean_terminated_length": 1157.83642578125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2666237447410822, "epoch": 0.1054806483907385, "frac_reward_zero_std": 0.375, "grad_norm": 0.15779152512550354, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 268457326.0, "reward": 0.625, "reward_std": 0.2566280663013458, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 619, "tools/generated_tokens": 4231.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1398.3671875, "completions/mean_terminated_length": 1199.5101318359375, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.28670331183820963, "epoch": 0.10565105331544082, "frac_reward_zero_std": 0.25, "grad_norm": 0.18112021684646606, "learning_rate": 1e-06, "loss": 0.0224, "num_tokens": 268896060.0, "reward": 0.50390625, "reward_std": 0.31625896692276, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 620, "tools/generated_tokens": 4774.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1314.05078125, "completions/mean_terminated_length": 1161.7264404296875, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.30507533717900515, "epoch": 0.10582145824014313, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1667071133852005, "learning_rate": 1e-06, "loss": 0.0215, "num_tokens": 269318681.0, "reward": 0.546875, "reward_std": 0.2688092887401581, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 621, "tools/generated_tokens": 4810.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.70703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1413.1328125, "completions/mean_terminated_length": 1284.9765625, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.23536482453346252, "epoch": 0.10599186316484546, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14880546927452087, "learning_rate": 1e-06, "loss": 0.0118, "num_tokens": 269755499.0, "reward": 0.4921875, "reward_std": 0.21251130104064941, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 622, "tools/generated_tokens": 4341.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1376.671875, "completions/mean_terminated_length": 1248.651123046875, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.2502811774611473, "epoch": 0.10616226808954779, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1514146625995636, "learning_rate": 1e-06, "loss": 0.0182, "num_tokens": 270193735.0, "reward": 0.51953125, "reward_std": 0.24031278491020203, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 623, "tools/generated_tokens": 4480.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1443.97265625, "completions/mean_terminated_length": 1188.93896484375, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.31538047548383474, "epoch": 0.10633267301425012, "frac_reward_zero_std": 0.5, "grad_norm": 0.16263020038604736, "learning_rate": 1e-06, "loss": -0.0018, "num_tokens": 270643968.0, "reward": 0.38671875, "reward_std": 0.18760645389556885, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 624, "tools/generated_tokens": 5195.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.83203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1451.19140625, "completions/mean_terminated_length": 1327.325439453125, "completions/min_length": 292.0, "completions/min_terminated_length": 292.0, "entropy": 0.24140130449086428, "epoch": 0.10650307793895243, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1506175845861435, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 271075569.0, "reward": 0.59765625, "reward_std": 0.26320207118988037, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 625, "tools/generated_tokens": 3891.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1379.875, "completions/mean_terminated_length": 1213.6732177734375, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.2595217255875468, "epoch": 0.10667348286365476, "frac_reward_zero_std": 0.625, "grad_norm": 0.09836837649345398, "learning_rate": 1e-06, "loss": -0.0036, "num_tokens": 271511185.0, "reward": 0.38671875, "reward_std": 0.12742365896701813, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 626, "tools/generated_tokens": 4803.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1334.44921875, "completions/mean_terminated_length": 1156.9366455078125, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.2846803767606616, "epoch": 0.10684388778835709, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20360346138477325, "learning_rate": 1e-06, "loss": 0.0557, "num_tokens": 271942484.0, "reward": 0.52734375, "reward_std": 0.25221139192581177, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 627, "tools/generated_tokens": 4574.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1323.11328125, "completions/mean_terminated_length": 1147.1795654296875, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.28598304837942123, "epoch": 0.10701429271305941, "frac_reward_zero_std": 0.5, "grad_norm": 0.13000962138175964, "learning_rate": 1e-06, "loss": 0.0366, "num_tokens": 272365921.0, "reward": 0.55859375, "reward_std": 0.22273029386997223, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 628, "tools/generated_tokens": 4563.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1264.2421875, "completions/mean_terminated_length": 1140.11767578125, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.26456060726195574, "epoch": 0.10718469763776173, "frac_reward_zero_std": 0.1875, "grad_norm": 0.18791763484477997, "learning_rate": 1e-06, "loss": 0.0556, "num_tokens": 272782175.0, "reward": 0.69140625, "reward_std": 0.29092884063720703, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 629, "tools/generated_tokens": 4424.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 1994.0, "completions/mean_length": 1219.8359375, "completions/mean_terminated_length": 1088.6788330078125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2776689175516367, "epoch": 0.10735510256246406, "frac_reward_zero_std": 0.375, "grad_norm": 0.18204987049102783, "learning_rate": 1e-06, "loss": 0.0126, "num_tokens": 273177717.0, "reward": 0.484375, "reward_std": 0.2574812173843384, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 630, "tools/generated_tokens": 4147.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1435.2578125, "completions/mean_terminated_length": 1255.7677001953125, "completions/min_length": 221.0, "completions/min_terminated_length": 221.0, "entropy": 0.23397820256650448, "epoch": 0.10752550748716638, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13757413625717163, "learning_rate": 1e-06, "loss": 0.0312, "num_tokens": 273619207.0, "reward": 0.42578125, "reward_std": 0.2495477795600891, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 631, "tools/generated_tokens": 4547.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1362.64453125, "completions/mean_terminated_length": 1261.228759765625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21338557358831167, "epoch": 0.1076959124118687, "frac_reward_zero_std": 0.5625, "grad_norm": 0.13186493515968323, "learning_rate": 1e-06, "loss": 0.0135, "num_tokens": 274036140.0, "reward": 0.70703125, "reward_std": 0.1438203752040863, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 632, "tools/generated_tokens": 3578.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.08203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1378.7265625, "completions/mean_terminated_length": 1164.8349609375, "completions/min_length": 271.0, "completions/min_terminated_length": 271.0, "entropy": 0.2828444391489029, "epoch": 0.10786631733657102, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1869775801897049, "learning_rate": 1e-06, "loss": 0.0254, "num_tokens": 274472630.0, "reward": 0.40625, "reward_std": 0.2245136797428131, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 633, "tools/generated_tokens": 4930.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1325.94140625, "completions/mean_terminated_length": 1114.4293212890625, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.32327297516167164, "epoch": 0.10803672226127335, "frac_reward_zero_std": 0.25, "grad_norm": 0.2007516771554947, "learning_rate": 1e-06, "loss": 0.0409, "num_tokens": 274901815.0, "reward": 0.44140625, "reward_std": 0.34267544746398926, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 634, "tools/generated_tokens": 5349.953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1380.5078125, "completions/mean_terminated_length": 1226.485595703125, "completions/min_length": 345.0, "completions/min_terminated_length": 345.0, "entropy": 0.2673746030777693, "epoch": 0.10820712718597568, "frac_reward_zero_std": 0.5, "grad_norm": 0.1623433232307434, "learning_rate": 1e-06, "loss": 0.0131, "num_tokens": 275328969.0, "reward": 0.39453125, "reward_std": 0.18265536427497864, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 635, "tools/generated_tokens": 4452.53125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1424.09375, "completions/mean_terminated_length": 1145.6328125, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.3063347237184644, "epoch": 0.108377532110678, "frac_reward_zero_std": 0.5625, "grad_norm": 0.11974193900823593, "learning_rate": 1e-06, "loss": 0.0016, "num_tokens": 275771329.0, "reward": 0.3671875, "reward_std": 0.2048833966255188, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 636, "tools/generated_tokens": 5112.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1169.5, "completions/mean_terminated_length": 1118.6776123046875, "completions/min_length": 161.0, "completions/min_terminated_length": 161.0, "entropy": 0.27197619155049324, "epoch": 0.10854793703538032, "frac_reward_zero_std": 0.5, "grad_norm": 0.18804891407489777, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 276148433.0, "reward": 0.6484375, "reward_std": 0.1624118983745575, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 637, "tools/generated_tokens": 3609.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1482.41796875, "completions/mean_terminated_length": 1215.8792724609375, "completions/min_length": 308.0, "completions/min_terminated_length": 308.0, "entropy": 0.2589757265523076, "epoch": 0.10871834196008265, "frac_reward_zero_std": 0.625, "grad_norm": 0.15880268812179565, "learning_rate": 1e-06, "loss": -0.0156, "num_tokens": 276618396.0, "reward": 0.4609375, "reward_std": 0.13041725754737854, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 638, "tools/generated_tokens": 5226.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1399.01953125, "completions/mean_terminated_length": 1204.659912109375, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.28624267783015966, "epoch": 0.10888874688478498, "frac_reward_zero_std": 0.1875, "grad_norm": 0.171358123421669, "learning_rate": 1e-06, "loss": 0.0569, "num_tokens": 277053121.0, "reward": 0.52734375, "reward_std": 0.29081130027770996, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 639, "tools/generated_tokens": 4807.01953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1357.08203125, "completions/mean_terminated_length": 1209.7298583984375, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.2540802387520671, "epoch": 0.10905915180948729, "frac_reward_zero_std": 0.375, "grad_norm": 0.14097262918949127, "learning_rate": 1e-06, "loss": 0.0199, "num_tokens": 277480694.0, "reward": 0.57421875, "reward_std": 0.2199878990650177, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 640, "tools/generated_tokens": 4229.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1367.78515625, "completions/mean_terminated_length": 1121.7552490234375, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.29023122135549784, "epoch": 0.10922955673418962, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14141590893268585, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 277909999.0, "reward": 0.5703125, "reward_std": 0.16901493072509766, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 641, "tools/generated_tokens": 4863.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.70703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1321.83984375, "completions/mean_terminated_length": 1158.54541015625, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.2669211020693183, "epoch": 0.10939996165889194, "frac_reward_zero_std": 0.0625, "grad_norm": 0.1905251145362854, "learning_rate": 1e-06, "loss": 0.0207, "num_tokens": 278335174.0, "reward": 0.5859375, "reward_std": 0.38027530908584595, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 642, "tools/generated_tokens": 4841.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1393.015625, "completions/mean_terminated_length": 1222.0196533203125, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.28753375727683306, "epoch": 0.10957036658359427, "frac_reward_zero_std": 0.25, "grad_norm": 0.1880209594964981, "learning_rate": 1e-06, "loss": 0.0206, "num_tokens": 278784234.0, "reward": 0.53515625, "reward_std": 0.32094109058380127, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 643, "tools/generated_tokens": 5273.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.89453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1283.921875, "completions/mean_terminated_length": 1178.6533203125, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.3149437680840492, "epoch": 0.10974077150829659, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20286233723163605, "learning_rate": 1e-06, "loss": 0.0057, "num_tokens": 279199222.0, "reward": 0.3515625, "reward_std": 0.2900395393371582, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 644, "tools/generated_tokens": 4731.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1315.46484375, "completions/mean_terminated_length": 1128.7451171875, "completions/min_length": 347.0, "completions/min_terminated_length": 347.0, "entropy": 0.23532930668443441, "epoch": 0.10991117643299891, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16920538246631622, "learning_rate": 1e-06, "loss": 0.0417, "num_tokens": 279621165.0, "reward": 0.5625, "reward_std": 0.3102988600730896, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 645, "tools/generated_tokens": 4867.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1283.63671875, "completions/mean_terminated_length": 1189.767578125, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.2547568343579769, "epoch": 0.11008158135770124, "frac_reward_zero_std": 0.375, "grad_norm": 0.166712686419487, "learning_rate": 1e-06, "loss": -0.0026, "num_tokens": 280022832.0, "reward": 0.72265625, "reward_std": 0.2599048614501953, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 646, "tools/generated_tokens": 3891.65234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1382.25390625, "completions/mean_terminated_length": 1106.3978271484375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.2551482766866684, "epoch": 0.11025198628240356, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1152617335319519, "learning_rate": 1e-06, "loss": 0.0169, "num_tokens": 280451825.0, "reward": 0.36328125, "reward_std": 0.17473775148391724, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 647, "tools/generated_tokens": 4974.265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1468.33203125, "completions/mean_terminated_length": 1185.244140625, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.29337296821177006, "epoch": 0.11042239120710588, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15559512376785278, "learning_rate": 1e-06, "loss": 0.0193, "num_tokens": 280907798.0, "reward": 0.375, "reward_std": 0.22383463382720947, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 648, "tools/generated_tokens": 5316.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2014.0, "completions/mean_length": 1352.0234375, "completions/mean_terminated_length": 1199.5810546875, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.26699577923864126, "epoch": 0.11059279613180821, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1800403594970703, "learning_rate": 1e-06, "loss": -0.0008, "num_tokens": 281332316.0, "reward": 0.37890625, "reward_std": 0.29315072298049927, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 649, "tools/generated_tokens": 4584.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1403.3671875, "completions/mean_terminated_length": 1197.36083984375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.3092813640832901, "epoch": 0.11076320105651054, "frac_reward_zero_std": 0.125, "grad_norm": 0.22655443847179413, "learning_rate": 1e-06, "loss": 0.007, "num_tokens": 281785338.0, "reward": 0.4140625, "reward_std": 0.34595030546188354, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 650, "tools/generated_tokens": 5331.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1372.65234375, "completions/mean_terminated_length": 1243.8651123046875, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.24570453632622957, "epoch": 0.11093360598121285, "frac_reward_zero_std": 0.125, "grad_norm": 0.19229960441589355, "learning_rate": 1e-06, "loss": 0.0379, "num_tokens": 282206161.0, "reward": 0.4296875, "reward_std": 0.3614438474178314, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 651, "tools/generated_tokens": 4068.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1399.1953125, "completions/mean_terminated_length": 1271.869140625, "completions/min_length": 242.0, "completions/min_terminated_length": 242.0, "entropy": 0.2655220804736018, "epoch": 0.11110401090591518, "frac_reward_zero_std": 0.375, "grad_norm": 0.2009373903274536, "learning_rate": 1e-06, "loss": 0.0113, "num_tokens": 282644275.0, "reward": 0.671875, "reward_std": 0.25789541006088257, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 652, "tools/generated_tokens": 4471.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1279.20703125, "completions/mean_terminated_length": 1157.4525146484375, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.2848825789988041, "epoch": 0.1112744158306175, "frac_reward_zero_std": 0.3125, "grad_norm": 0.45934826135635376, "learning_rate": 1e-06, "loss": 0.0318, "num_tokens": 283052248.0, "reward": 0.62109375, "reward_std": 0.2815985083580017, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 653, "tools/generated_tokens": 4223.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1350.4375, "completions/mean_terminated_length": 1146.1162109375, "completions/min_length": 215.0, "completions/min_terminated_length": 215.0, "entropy": 0.2441366296261549, "epoch": 0.11144482075531983, "frac_reward_zero_std": 0.25, "grad_norm": 0.19023865461349487, "learning_rate": 1e-06, "loss": -0.0051, "num_tokens": 283476136.0, "reward": 0.6328125, "reward_std": 0.25418204069137573, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 654, "tools/generated_tokens": 4622.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1433.6328125, "completions/mean_terminated_length": 1188.557373046875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2219137530773878, "epoch": 0.11161522568002215, "frac_reward_zero_std": 0.25, "grad_norm": 0.16890238225460052, "learning_rate": 1e-06, "loss": 0.037, "num_tokens": 283929194.0, "reward": 0.4375, "reward_std": 0.31049323081970215, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 655, "tools/generated_tokens": 5113.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1378.3828125, "completions/mean_terminated_length": 1150.502685546875, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.3080749027431011, "epoch": 0.11178563060472448, "frac_reward_zero_std": 0.1875, "grad_norm": 0.19281238317489624, "learning_rate": 1e-06, "loss": 0.0428, "num_tokens": 284374156.0, "reward": 0.4765625, "reward_std": 0.32010379433631897, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 656, "tools/generated_tokens": 5298.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2011.0, "completions/mean_length": 1335.77734375, "completions/mean_terminated_length": 1029.4022216796875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.28125489316880703, "epoch": 0.1119560355294268, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14651690423488617, "learning_rate": 1e-06, "loss": 0.0199, "num_tokens": 284806371.0, "reward": 0.3125, "reward_std": 0.16406384110450745, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 657, "tools/generated_tokens": 5279.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1250.56640625, "completions/mean_terminated_length": 1156.5458984375, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.2504276493564248, "epoch": 0.11212644045412913, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15595729649066925, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 285204836.0, "reward": 0.57421875, "reward_std": 0.2115791141986847, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 658, "tools/generated_tokens": 3930.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1384.0, "completions/mean_terminated_length": 1148.61376953125, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.2641846025362611, "epoch": 0.11229684537883144, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17719745635986328, "learning_rate": 1e-06, "loss": 0.052, "num_tokens": 285653396.0, "reward": 0.4140625, "reward_std": 0.29392436146736145, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 659, "tools/generated_tokens": 5304.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.32421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1478.26953125, "completions/mean_terminated_length": 1204.9364013671875, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.2594412565231323, "epoch": 0.11246725030353377, "frac_reward_zero_std": 0.625, "grad_norm": 0.1273634135723114, "learning_rate": 1e-06, "loss": 0.0096, "num_tokens": 286108489.0, "reward": 0.46484375, "reward_std": 0.15309548377990723, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 660, "tools/generated_tokens": 4774.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1349.515625, "completions/mean_terminated_length": 1188.331787109375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.26714857015758753, "epoch": 0.1126376552282361, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18857212364673615, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 286532477.0, "reward": 0.39453125, "reward_std": 0.27077996730804443, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 661, "tools/generated_tokens": 4765.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1258.6484375, "completions/mean_terminated_length": 1137.7657470703125, "completions/min_length": 189.0, "completions/min_terminated_length": 189.0, "entropy": 0.2908195350319147, "epoch": 0.11280806015293841, "frac_reward_zero_std": 0.25, "grad_norm": 0.2098398655653, "learning_rate": 1e-06, "loss": 0.0147, "num_tokens": 286939363.0, "reward": 0.40234375, "reward_std": 0.28387558460235596, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 662, "tools/generated_tokens": 4466.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1212.98046875, "completions/mean_terminated_length": 1122.61474609375, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.2730645714327693, "epoch": 0.11297846507764074, "frac_reward_zero_std": 0.375, "grad_norm": 0.18227653205394745, "learning_rate": 1e-06, "loss": 0.0048, "num_tokens": 287322398.0, "reward": 0.55859375, "reward_std": 0.26917997002601624, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 663, "tools/generated_tokens": 3732.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1478.234375, "completions/mean_terminated_length": 1268.00537109375, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "entropy": 0.27151405811309814, "epoch": 0.11314887000234307, "frac_reward_zero_std": 0.3125, "grad_norm": 0.30221056938171387, "learning_rate": 1e-06, "loss": 0.034, "num_tokens": 287789690.0, "reward": 0.3671875, "reward_std": 0.263522744178772, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 664, "tools/generated_tokens": 5406.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1293.9765625, "completions/mean_terminated_length": 1128.8095703125, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.27544736210256815, "epoch": 0.1133192749270454, "frac_reward_zero_std": 0.3125, "grad_norm": 0.6790370941162109, "learning_rate": 1e-06, "loss": 0.0258, "num_tokens": 288210468.0, "reward": 0.46484375, "reward_std": 0.29674431681632996, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 665, "tools/generated_tokens": 4933.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1331.7265625, "completions/mean_terminated_length": 1097.9171142578125, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.23813448939472437, "epoch": 0.11348967985174771, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15418995916843414, "learning_rate": 1e-06, "loss": 0.0461, "num_tokens": 288640190.0, "reward": 0.6328125, "reward_std": 0.2523040473461151, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 666, "tools/generated_tokens": 4843.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1196.55859375, "completions/mean_terminated_length": 1087.7840576171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.3058233577758074, "epoch": 0.11366008477645004, "frac_reward_zero_std": 0.5, "grad_norm": 0.15078438818454742, "learning_rate": 1e-06, "loss": 0.0207, "num_tokens": 289018029.0, "reward": 0.5390625, "reward_std": 0.19994549453258514, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 667, "tools/generated_tokens": 3532.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1408.40234375, "completions/mean_terminated_length": 1221.050537109375, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.2560304347425699, "epoch": 0.11383048970115237, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1727852076292038, "learning_rate": 1e-06, "loss": 0.0189, "num_tokens": 289479988.0, "reward": 0.3984375, "reward_std": 0.30986616015434265, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 668, "tools/generated_tokens": 5016.4140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1272.8359375, "completions/mean_terminated_length": 1098.5167236328125, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.24627330992370844, "epoch": 0.11400089462585469, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14643415808677673, "learning_rate": 1e-06, "loss": 0.0214, "num_tokens": 289885786.0, "reward": 0.5625, "reward_std": 0.2265685796737671, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 669, "tools/generated_tokens": 4368.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1236.69140625, "completions/mean_terminated_length": 1137.0745849609375, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.27028775587677956, "epoch": 0.114171299550557, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14379100501537323, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 290281147.0, "reward": 0.7109375, "reward_std": 0.21433347463607788, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 670, "tools/generated_tokens": 4028.703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1269.59375, "completions/mean_terminated_length": 1108.0377197265625, "completions/min_length": 293.0, "completions/min_terminated_length": 293.0, "entropy": 0.25096935499459505, "epoch": 0.11434170447525933, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15770353376865387, "learning_rate": 1e-06, "loss": 0.032, "num_tokens": 290695283.0, "reward": 0.55078125, "reward_std": 0.22028236091136932, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 671, "tools/generated_tokens": 4477.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1248.41796875, "completions/mean_terminated_length": 1087.0, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.2772094663232565, "epoch": 0.11451210939996166, "frac_reward_zero_std": 0.375, "grad_norm": 0.1849849820137024, "learning_rate": 1e-06, "loss": 0.0025, "num_tokens": 291104094.0, "reward": 0.40625, "reward_std": 0.26416581869125366, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 672, "tools/generated_tokens": 4712.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1195.19921875, "completions/mean_terminated_length": 1094.650634765625, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.2656153868883848, "epoch": 0.11468251432466399, "frac_reward_zero_std": 0.1875, "grad_norm": 0.18833574652671814, "learning_rate": 1e-06, "loss": -0.0117, "num_tokens": 291495249.0, "reward": 0.52734375, "reward_std": 0.300828218460083, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 673, "tools/generated_tokens": 4363.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1407.12109375, "completions/mean_terminated_length": 1175.3138427734375, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.2654810417443514, "epoch": 0.1148529192493663, "frac_reward_zero_std": 0.25, "grad_norm": 0.18689681589603424, "learning_rate": 1e-06, "loss": 0.0369, "num_tokens": 291934592.0, "reward": 0.453125, "reward_std": 0.3470836579799652, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 674, "tools/generated_tokens": 4823.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1353.97265625, "completions/mean_terminated_length": 1159.64501953125, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.2787818741053343, "epoch": 0.11502332417406863, "frac_reward_zero_std": 0.375, "grad_norm": 0.15539275109767914, "learning_rate": 1e-06, "loss": 0.0198, "num_tokens": 292371289.0, "reward": 0.53515625, "reward_std": 0.23942145705223083, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 675, "tools/generated_tokens": 5033.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1319.33984375, "completions/mean_terminated_length": 1151.1971435546875, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.2657048776745796, "epoch": 0.11519372909877096, "frac_reward_zero_std": 0.375, "grad_norm": 0.18021747469902039, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 292791952.0, "reward": 0.40234375, "reward_std": 0.260199636220932, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 676, "tools/generated_tokens": 4383.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1340.1484375, "completions/mean_terminated_length": 1176.8173828125, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.21105117443948984, "epoch": 0.11536413402347327, "frac_reward_zero_std": 0.25, "grad_norm": 0.16563092172145844, "learning_rate": 1e-06, "loss": 0.0177, "num_tokens": 293213606.0, "reward": 0.62109375, "reward_std": 0.30369094014167786, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 677, "tools/generated_tokens": 4212.1640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1331.52734375, "completions/mean_terminated_length": 1148.9019775390625, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2592040905728936, "epoch": 0.1155345389481756, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17642486095428467, "learning_rate": 1e-06, "loss": 0.0382, "num_tokens": 293634125.0, "reward": 0.59375, "reward_std": 0.23083871603012085, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 678, "tools/generated_tokens": 4603.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1387.69921875, "completions/mean_terminated_length": 1194.302978515625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.23489708360284567, "epoch": 0.11570494387287793, "frac_reward_zero_std": 0.625, "grad_norm": 0.13390317559242249, "learning_rate": 1e-06, "loss": 0.0317, "num_tokens": 294065840.0, "reward": 0.4609375, "reward_std": 0.16470219194889069, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 679, "tools/generated_tokens": 4459.7421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1468.8046875, "completions/mean_terminated_length": 1195.8563232421875, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.2739240461960435, "epoch": 0.11587534879758025, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17326202988624573, "learning_rate": 1e-06, "loss": 0.0209, "num_tokens": 294518750.0, "reward": 0.45703125, "reward_std": 0.2021270990371704, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 680, "tools/generated_tokens": 5108.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1340.69140625, "completions/mean_terminated_length": 1169.0145263671875, "completions/min_length": 355.0, "completions/min_terminated_length": 355.0, "entropy": 0.27234411612153053, "epoch": 0.11604575372228257, "frac_reward_zero_std": 0.5, "grad_norm": 0.18925684690475464, "learning_rate": 1e-06, "loss": 0.0357, "num_tokens": 294951855.0, "reward": 0.46484375, "reward_std": 0.19278642535209656, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 681, "tools/generated_tokens": 4924.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1425.78515625, "completions/mean_terminated_length": 1333.717529296875, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.21625436283648014, "epoch": 0.1162161586469849, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1252322494983673, "learning_rate": 1e-06, "loss": 0.0216, "num_tokens": 295381896.0, "reward": 0.75390625, "reward_std": 0.22438223659992218, "rewards/simpleverify_reward/mean": 0.75390625, "rewards/simpleverify_reward/std": 0.43157756328582764, "step": 682, "tools/generated_tokens": 3857.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1298.3671875, "completions/mean_terminated_length": 1142.7877197265625, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.22653070464730263, "epoch": 0.11638656357168722, "frac_reward_zero_std": 0.3125, "grad_norm": 0.15767717361450195, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 295794966.0, "reward": 0.56640625, "reward_std": 0.27572914958000183, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 683, "tools/generated_tokens": 4322.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1334.33203125, "completions/mean_terminated_length": 1156.804931640625, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.24402422830462456, "epoch": 0.11655696849638955, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15608720481395721, "learning_rate": 1e-06, "loss": 0.0107, "num_tokens": 296211579.0, "reward": 0.48828125, "reward_std": 0.2114706039428711, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 684, "tools/generated_tokens": 4678.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1433.1484375, "completions/mean_terminated_length": 1122.12353515625, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.2377507919445634, "epoch": 0.11672737342109187, "frac_reward_zero_std": 0.3125, "grad_norm": 0.21769921481609344, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 296663313.0, "reward": 0.45703125, "reward_std": 0.2530073821544647, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 685, "tools/generated_tokens": 5337.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.90625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1391.33984375, "completions/mean_terminated_length": 1163.2369384765625, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.2597609106451273, "epoch": 0.11689777834579419, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16523295640945435, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 297101864.0, "reward": 0.51953125, "reward_std": 0.22704584896564484, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 686, "tools/generated_tokens": 4775.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1502.62109375, "completions/mean_terminated_length": 1280.8846435546875, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.24336642771959305, "epoch": 0.11706818327049652, "frac_reward_zero_std": 0.5, "grad_norm": 0.14157827198505402, "learning_rate": 1e-06, "loss": 0.0157, "num_tokens": 297562263.0, "reward": 0.49609375, "reward_std": 0.19090843200683594, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 687, "tools/generated_tokens": 4734.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1234.9375, "completions/mean_terminated_length": 1070.8028564453125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.25796001125127077, "epoch": 0.11723858819519885, "frac_reward_zero_std": 0.5, "grad_norm": 0.18083597719669342, "learning_rate": 1e-06, "loss": 0.0215, "num_tokens": 297953623.0, "reward": 0.41015625, "reward_std": 0.2173290103673935, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 688, "tools/generated_tokens": 4298.9453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1457.578125, "completions/mean_terminated_length": 1194.073486328125, "completions/min_length": 255.0, "completions/min_terminated_length": 255.0, "entropy": 0.2565632164478302, "epoch": 0.11740899311990116, "frac_reward_zero_std": 0.125, "grad_norm": 0.1800205558538437, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 298415019.0, "reward": 0.578125, "reward_std": 0.29830074310302734, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 689, "tools/generated_tokens": 5321.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.88671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1353.52734375, "completions/mean_terminated_length": 1140.938720703125, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.25891492888331413, "epoch": 0.11757939804460349, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15302583575248718, "learning_rate": 1e-06, "loss": 0.034, "num_tokens": 298848098.0, "reward": 0.42578125, "reward_std": 0.16415652632713318, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 690, "tools/generated_tokens": 4897.52734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1498.70703125, "completions/mean_terminated_length": 1249.0284423828125, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.2533171446993947, "epoch": 0.11774980296930582, "frac_reward_zero_std": 0.375, "grad_norm": 0.13147640228271484, "learning_rate": 1e-06, "loss": 0.0304, "num_tokens": 299305415.0, "reward": 0.41796875, "reward_std": 0.24447914958000183, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 691, "tools/generated_tokens": 4906.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1342.39453125, "completions/mean_terminated_length": 1162.563720703125, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.20845069084316492, "epoch": 0.11792020789400813, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14937786757946014, "learning_rate": 1e-06, "loss": -0.0029, "num_tokens": 299720140.0, "reward": 0.47265625, "reward_std": 0.2062118798494339, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 692, "tools/generated_tokens": 4318.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1480.01171875, "completions/mean_terminated_length": 1257.771728515625, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.20788600947707891, "epoch": 0.11809061281871046, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1496376246213913, "learning_rate": 1e-06, "loss": 0.0196, "num_tokens": 300176031.0, "reward": 0.58203125, "reward_std": 0.2978776693344116, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 693, "tools/generated_tokens": 4904.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1348.3359375, "completions/mean_terminated_length": 1090.171142578125, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.3026274088770151, "epoch": 0.11826101774341279, "frac_reward_zero_std": 0.375, "grad_norm": 0.17613011598587036, "learning_rate": 1e-06, "loss": 0.0267, "num_tokens": 300613333.0, "reward": 0.34765625, "reward_std": 0.2727735638618469, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 694, "tools/generated_tokens": 5364.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1400.03125, "completions/mean_terminated_length": 1222.7313232421875, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 0.26990509778261185, "epoch": 0.11843142266811511, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17711031436920166, "learning_rate": 1e-06, "loss": 0.0252, "num_tokens": 301056621.0, "reward": 0.53125, "reward_std": 0.21864622831344604, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 695, "tools/generated_tokens": 5104.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2012.0, "completions/mean_length": 1370.99609375, "completions/mean_terminated_length": 1095.75830078125, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.23035681061446667, "epoch": 0.11860182759281743, "frac_reward_zero_std": 0.5, "grad_norm": 0.1524369716644287, "learning_rate": 1e-06, "loss": 0.0249, "num_tokens": 301494636.0, "reward": 0.5, "reward_std": 0.19760414958000183, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 696, "tools/generated_tokens": 4899.015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1338.0625, "completions/mean_terminated_length": 1148.2772216796875, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.23033000621944666, "epoch": 0.11877223251751975, "frac_reward_zero_std": 0.0625, "grad_norm": 0.22331736981868744, "learning_rate": 1e-06, "loss": 0.0292, "num_tokens": 301923404.0, "reward": 0.59765625, "reward_std": 0.41821908950805664, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 697, "tools/generated_tokens": 4842.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1406.28125, "completions/mean_terminated_length": 1261.97119140625, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.22601748164743185, "epoch": 0.11894263744222208, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15356329083442688, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 302363300.0, "reward": 0.453125, "reward_std": 0.22843991219997406, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 698, "tools/generated_tokens": 4582.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1260.859375, "completions/mean_terminated_length": 1168.0523681640625, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.24092142656445503, "epoch": 0.11911304236692441, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1545473337173462, "learning_rate": 1e-06, "loss": -0.0011, "num_tokens": 302758256.0, "reward": 0.61328125, "reward_std": 0.2468073070049286, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 699, "tools/generated_tokens": 3692.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1433.78515625, "completions/mean_terminated_length": 1184.054931640625, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.25013092439621687, "epoch": 0.11928344729162672, "frac_reward_zero_std": 0.4375, "grad_norm": 0.13353323936462402, "learning_rate": 1e-06, "loss": 0.017, "num_tokens": 303212809.0, "reward": 0.5078125, "reward_std": 0.21973668038845062, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 700, "tools/generated_tokens": 5297.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.88671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1309.0078125, "completions/mean_terminated_length": 1092.535400390625, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.2741047888994217, "epoch": 0.11945385221632905, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2745325565338135, "learning_rate": 1e-06, "loss": 0.0262, "num_tokens": 303633243.0, "reward": 0.23046875, "reward_std": 0.22974532842636108, "rewards/simpleverify_reward/mean": 0.23046875, "rewards/simpleverify_reward/std": 0.4219578504562378, "step": 701, "tools/generated_tokens": 5077.01953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.83984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1444.83203125, "completions/mean_terminated_length": 1139.7353515625, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.2707134699448943, "epoch": 0.11962425714103138, "frac_reward_zero_std": 0.5, "grad_norm": 0.16257143020629883, "learning_rate": 1e-06, "loss": 0.0182, "num_tokens": 304077776.0, "reward": 0.4375, "reward_std": 0.21388331055641174, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 702, "tools/generated_tokens": 4956.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1345.25, "completions/mean_terminated_length": 1244.87060546875, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.21007448062300682, "epoch": 0.1197946620657337, "frac_reward_zero_std": 0.5, "grad_norm": 0.13849498331546783, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 304489776.0, "reward": 0.609375, "reward_std": 0.2241290807723999, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 703, "tools/generated_tokens": 3713.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1321.91015625, "completions/mean_terminated_length": 1179.4111328125, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.21439216658473015, "epoch": 0.11996506699043602, "frac_reward_zero_std": 0.8125, "grad_norm": 0.0818270817399025, "learning_rate": 1e-06, "loss": 0.008, "num_tokens": 304891865.0, "reward": 0.61328125, "reward_std": 0.09122256934642792, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 704, "tools/generated_tokens": 3377.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.00390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1415.1953125, "completions/mean_terminated_length": 1242.0447998046875, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.2485204804688692, "epoch": 0.12013547191513835, "frac_reward_zero_std": 0.375, "grad_norm": 0.14054827392101288, "learning_rate": 1e-06, "loss": 0.0321, "num_tokens": 305346059.0, "reward": 0.56640625, "reward_std": 0.25507354736328125, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 705, "tools/generated_tokens": 4535.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1341.36328125, "completions/mean_terminated_length": 1095.9105224609375, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.25401805620640516, "epoch": 0.12030587683984068, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15206822752952576, "learning_rate": 1e-06, "loss": 0.0322, "num_tokens": 305778264.0, "reward": 0.5, "reward_std": 0.22327595949172974, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 706, "tools/generated_tokens": 4981.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1274.87890625, "completions/mean_terminated_length": 1096.4759521484375, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.2554048392921686, "epoch": 0.12047628176454299, "frac_reward_zero_std": 0.5, "grad_norm": 0.14919979870319366, "learning_rate": 1e-06, "loss": 0.0104, "num_tokens": 306180441.0, "reward": 0.47265625, "reward_std": 0.20223368704319, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 707, "tools/generated_tokens": 4170.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1278.359375, "completions/mean_terminated_length": 1168.4107666015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.259488970041275, "epoch": 0.12064668668924532, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15932126343250275, "learning_rate": 1e-06, "loss": 0.0098, "num_tokens": 306576677.0, "reward": 0.5859375, "reward_std": 0.16713693737983704, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 708, "tools/generated_tokens": 3806.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1360.66015625, "completions/mean_terminated_length": 1064.98876953125, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2471571397036314, "epoch": 0.12081709161394764, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17095567286014557, "learning_rate": 1e-06, "loss": 0.0218, "num_tokens": 307009278.0, "reward": 0.484375, "reward_std": 0.20851992070674896, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 709, "tools/generated_tokens": 5192.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1168.7421875, "completions/mean_terminated_length": 976.1571655273438, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.24509234726428986, "epoch": 0.12098749653864997, "frac_reward_zero_std": 0.625, "grad_norm": 0.12865740060806274, "learning_rate": 1e-06, "loss": -0.0094, "num_tokens": 307387116.0, "reward": 0.49609375, "reward_std": 0.13039018213748932, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 710, "tools/generated_tokens": 4024.7578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1396.953125, "completions/mean_terminated_length": 1188.8917236328125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.26750652492046356, "epoch": 0.12115790146335229, "frac_reward_zero_std": 0.1875, "grad_norm": 0.19605718553066254, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 307823776.0, "reward": 0.5546875, "reward_std": 0.3256245255470276, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 711, "tools/generated_tokens": 4684.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1311.0234375, "completions/mean_terminated_length": 1162.244140625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.25421780720353127, "epoch": 0.12132830638805461, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1831713616847992, "learning_rate": 1e-06, "loss": 0.0263, "num_tokens": 308240918.0, "reward": 0.58984375, "reward_std": 0.23776951432228088, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 712, "tools/generated_tokens": 4343.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1356.37890625, "completions/mean_terminated_length": 1158.2813720703125, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.25517632253468037, "epoch": 0.12149871131275694, "frac_reward_zero_std": 0.5, "grad_norm": 0.14055019617080688, "learning_rate": 1e-06, "loss": 0.0246, "num_tokens": 308660855.0, "reward": 0.546875, "reward_std": 0.18968652188777924, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 713, "tools/generated_tokens": 4332.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1404.41796875, "completions/mean_terminated_length": 1198.7421875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.24821795243769884, "epoch": 0.12166911623745927, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1955830156803131, "learning_rate": 1e-06, "loss": 0.0069, "num_tokens": 309103378.0, "reward": 0.51953125, "reward_std": 0.260877788066864, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 714, "tools/generated_tokens": 4692.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 1353.8984375, "completions/mean_terminated_length": 1136.769287109375, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.24956323858350515, "epoch": 0.12183952116216158, "frac_reward_zero_std": 0.25, "grad_norm": 0.26573148369789124, "learning_rate": 1e-06, "loss": 0.0268, "num_tokens": 309527944.0, "reward": 0.4609375, "reward_std": 0.275407612323761, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 715, "tools/generated_tokens": 4561.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.36328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1445.375, "completions/mean_terminated_length": 1101.57666015625, "completions/min_length": 258.0, "completions/min_terminated_length": 258.0, "entropy": 0.27512555941939354, "epoch": 0.12200992608686391, "frac_reward_zero_std": 0.4375, "grad_norm": 0.141936793923378, "learning_rate": 1e-06, "loss": 0.0286, "num_tokens": 309988312.0, "reward": 0.41796875, "reward_std": 0.23292973637580872, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 716, "tools/generated_tokens": 5589.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1284.85546875, "completions/mean_terminated_length": 1143.532470703125, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "entropy": 0.24519119411706924, "epoch": 0.12218033101156624, "frac_reward_zero_std": 0.375, "grad_norm": 0.1790136843919754, "learning_rate": 1e-06, "loss": 0.0138, "num_tokens": 310399795.0, "reward": 0.53125, "reward_std": 0.25110703706741333, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 717, "tools/generated_tokens": 4508.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.33203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2012.0, "completions/mean_length": 1366.08984375, "completions/mean_terminated_length": 1027.140380859375, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.27409070543944836, "epoch": 0.12235073593626856, "frac_reward_zero_std": 0.5, "grad_norm": 0.15358015894889832, "learning_rate": 1e-06, "loss": 0.0296, "num_tokens": 310837130.0, "reward": 0.3828125, "reward_std": 0.189048171043396, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 718, "tools/generated_tokens": 5374.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1412.98828125, "completions/mean_terminated_length": 1129.5762939453125, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.24863294791430235, "epoch": 0.12252114086097088, "frac_reward_zero_std": 0.375, "grad_norm": 0.9129813313484192, "learning_rate": 1e-06, "loss": 0.0326, "num_tokens": 311279415.0, "reward": 0.296875, "reward_std": 0.23339098691940308, "rewards/simpleverify_reward/mean": 0.296875, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 719, "tools/generated_tokens": 5149.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1457.515625, "completions/mean_terminated_length": 1221.9835205078125, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.2301028361544013, "epoch": 0.1226915457856732, "frac_reward_zero_std": 0.4375, "grad_norm": 0.14140602946281433, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 311734571.0, "reward": 0.578125, "reward_std": 0.24382495880126953, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 720, "tools/generated_tokens": 5281.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1244.66796875, "completions/mean_terminated_length": 1121.6396484375, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.2301520025357604, "epoch": 0.12286195071037553, "frac_reward_zero_std": 0.375, "grad_norm": 0.17017489671707153, "learning_rate": 1e-06, "loss": 0.0017, "num_tokens": 312135206.0, "reward": 0.43359375, "reward_std": 0.25967881083488464, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 721, "tools/generated_tokens": 4140.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1364.95703125, "completions/mean_terminated_length": 1160.390869140625, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "entropy": 0.22799314465373755, "epoch": 0.12303235563507785, "frac_reward_zero_std": 0.375, "grad_norm": 0.1625974029302597, "learning_rate": 1e-06, "loss": 0.0549, "num_tokens": 312562875.0, "reward": 0.40625, "reward_std": 0.2394353747367859, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 722, "tools/generated_tokens": 4492.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1318.16015625, "completions/mean_terminated_length": 1145.3961181640625, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.25876136031001806, "epoch": 0.12320276055978018, "frac_reward_zero_std": 0.3125, "grad_norm": 0.16170518100261688, "learning_rate": 1e-06, "loss": 0.0373, "num_tokens": 312981524.0, "reward": 0.578125, "reward_std": 0.25218653678894043, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 723, "tools/generated_tokens": 4622.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.61328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1262.41796875, "completions/mean_terminated_length": 1085.7559814453125, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.24038258753716946, "epoch": 0.1233731654844825, "frac_reward_zero_std": 0.25, "grad_norm": 0.189750537276268, "learning_rate": 1e-06, "loss": 0.0328, "num_tokens": 313394239.0, "reward": 0.53515625, "reward_std": 0.2747562527656555, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 724, "tools/generated_tokens": 4326.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1360.71484375, "completions/mean_terminated_length": 1102.0699462890625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.24909261241555214, "epoch": 0.12354357040918483, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18744732439517975, "learning_rate": 1e-06, "loss": 0.0272, "num_tokens": 313821094.0, "reward": 0.48046875, "reward_std": 0.2504550814628601, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 725, "tools/generated_tokens": 4976.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.32421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1441.70703125, "completions/mean_terminated_length": 1150.838134765625, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.22987970151007175, "epoch": 0.12371397533388714, "frac_reward_zero_std": 0.625, "grad_norm": 0.11026185005903244, "learning_rate": 1e-06, "loss": 0.0189, "num_tokens": 314274395.0, "reward": 0.44140625, "reward_std": 0.1156454086303711, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 726, "tools/generated_tokens": 5009.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1479.96875, "completions/mean_terminated_length": 1202.56396484375, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "entropy": 0.25264427438378334, "epoch": 0.12388438025858947, "frac_reward_zero_std": 0.3125, "grad_norm": 0.27290549874305725, "learning_rate": 1e-06, "loss": 0.0054, "num_tokens": 314741587.0, "reward": 0.3359375, "reward_std": 0.25947707891464233, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 727, "tools/generated_tokens": 5439.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1302.54296875, "completions/mean_terminated_length": 1098.5621337890625, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.2258697571232915, "epoch": 0.1240547851832918, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17060135304927826, "learning_rate": 1e-06, "loss": 0.0277, "num_tokens": 315158190.0, "reward": 0.56640625, "reward_std": 0.2657203674316406, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 728, "tools/generated_tokens": 4758.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1382.52734375, "completions/mean_terminated_length": 1169.8555908203125, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.2211061930283904, "epoch": 0.12422519010799413, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15268555283546448, "learning_rate": 1e-06, "loss": -0.0258, "num_tokens": 315585125.0, "reward": 0.3984375, "reward_std": 0.2252492606639862, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 729, "tools/generated_tokens": 4478.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1384.4921875, "completions/mean_terminated_length": 1215.36279296875, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.2283381512388587, "epoch": 0.12439559503269644, "frac_reward_zero_std": 0.5, "grad_norm": 0.15997079014778137, "learning_rate": 1e-06, "loss": 0.0464, "num_tokens": 316017027.0, "reward": 0.5703125, "reward_std": 0.23340700566768646, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 730, "tools/generated_tokens": 4664.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1342.6796875, "completions/mean_terminated_length": 1162.9019775390625, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.2390185883268714, "epoch": 0.12456599995739877, "frac_reward_zero_std": 0.3125, "grad_norm": 0.17097213864326477, "learning_rate": 1e-06, "loss": 0.0529, "num_tokens": 316447265.0, "reward": 0.50390625, "reward_std": 0.26264533400535583, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 731, "tools/generated_tokens": 4686.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1361.94921875, "completions/mean_terminated_length": 1207.674560546875, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2609061785042286, "epoch": 0.1247364048821011, "frac_reward_zero_std": 0.5, "grad_norm": 0.1683957874774933, "learning_rate": 1e-06, "loss": 0.014, "num_tokens": 316874436.0, "reward": 0.6328125, "reward_std": 0.1835355907678604, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 732, "tools/generated_tokens": 4409.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1302.30859375, "completions/mean_terminated_length": 1098.278564453125, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.19152087066322565, "epoch": 0.12490680980680342, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12962663173675537, "learning_rate": 1e-06, "loss": -0.0035, "num_tokens": 317289619.0, "reward": 0.44140625, "reward_std": 0.162959486246109, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 733, "tools/generated_tokens": 4286.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1373.1171875, "completions/mean_terminated_length": 1124.11767578125, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "entropy": 0.20564308110624552, "epoch": 0.12507721473150574, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1270165592432022, "learning_rate": 1e-06, "loss": 0.045, "num_tokens": 317724513.0, "reward": 0.5, "reward_std": 0.22941282391548157, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 734, "tools/generated_tokens": 4885.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1285.4765625, "completions/mean_terminated_length": 1067.0753173828125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23471161536872387, "epoch": 0.12524761965620806, "frac_reward_zero_std": 0.1875, "grad_norm": 0.18795357644557953, "learning_rate": 1e-06, "loss": 0.0143, "num_tokens": 318143227.0, "reward": 0.4921875, "reward_std": 0.299323707818985, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 735, "tools/generated_tokens": 4925.484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1204.640625, "completions/mean_terminated_length": 1029.603759765625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.26938064489513636, "epoch": 0.1254180245809104, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16767403483390808, "learning_rate": 1e-06, "loss": -0.0012, "num_tokens": 318540351.0, "reward": 0.53125, "reward_std": 0.2067594677209854, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 736, "tools/generated_tokens": 4772.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1412.6015625, "completions/mean_terminated_length": 1113.17236328125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.2551824441179633, "epoch": 0.12558842950561272, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1621071696281433, "learning_rate": 1e-06, "loss": 0.0234, "num_tokens": 318999049.0, "reward": 0.546875, "reward_std": 0.3011544942855835, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 737, "tools/generated_tokens": 5628.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.05859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1275.375, "completions/mean_terminated_length": 1119.3990478515625, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.2588097807019949, "epoch": 0.12575883443031505, "frac_reward_zero_std": 0.1875, "grad_norm": 0.215603768825531, "learning_rate": 1e-06, "loss": 0.0091, "num_tokens": 319412713.0, "reward": 0.5625, "reward_std": 0.32814085483551025, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 738, "tools/generated_tokens": 4635.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1366.0078125, "completions/mean_terminated_length": 1119.345703125, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.22428589407354593, "epoch": 0.12592923935501735, "frac_reward_zero_std": 0.125, "grad_norm": 0.1948491632938385, "learning_rate": 1e-06, "loss": 0.0425, "num_tokens": 319852811.0, "reward": 0.45703125, "reward_std": 0.3425579071044922, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 739, "tools/generated_tokens": 5510.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1233.390625, "completions/mean_terminated_length": 1133.350830078125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.25503019988536835, "epoch": 0.12609964427971967, "frac_reward_zero_std": 0.625, "grad_norm": 0.14445248246192932, "learning_rate": 1e-06, "loss": 0.0161, "num_tokens": 320237343.0, "reward": 0.37890625, "reward_std": 0.15338994562625885, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 740, "tools/generated_tokens": 3721.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1451.34765625, "completions/mean_terminated_length": 1280.46728515625, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.24542142823338509, "epoch": 0.126270049204422, "frac_reward_zero_std": 0.4375, "grad_norm": 0.15893565118312836, "learning_rate": 1e-06, "loss": 0.0127, "num_tokens": 320682744.0, "reward": 0.4765625, "reward_std": 0.2306618094444275, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 741, "tools/generated_tokens": 4715.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1330.66015625, "completions/mean_terminated_length": 1120.54541015625, "completions/min_length": 242.0, "completions/min_terminated_length": 242.0, "entropy": 0.26824986282736063, "epoch": 0.12644045412912433, "frac_reward_zero_std": 0.6875, "grad_norm": 0.12786538898944855, "learning_rate": 1e-06, "loss": 0.0266, "num_tokens": 321103169.0, "reward": 0.6015625, "reward_std": 0.1048629954457283, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 742, "tools/generated_tokens": 4434.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1232.00390625, "completions/mean_terminated_length": 1123.685791015625, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.2896163584664464, "epoch": 0.12661085905382666, "frac_reward_zero_std": 0.375, "grad_norm": 0.1879495531320572, "learning_rate": 1e-06, "loss": 0.0069, "num_tokens": 321500418.0, "reward": 0.4453125, "reward_std": 0.24466386437416077, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 743, "tools/generated_tokens": 4304.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1298.71484375, "completions/mean_terminated_length": 1093.696533203125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.20860141050070524, "epoch": 0.12678126397852899, "frac_reward_zero_std": 0.25, "grad_norm": 0.17947924137115479, "learning_rate": 1e-06, "loss": 0.0243, "num_tokens": 321921049.0, "reward": 0.5234375, "reward_std": 0.266292929649353, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 744, "tools/generated_tokens": 4794.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.70703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1431.578125, "completions/mean_terminated_length": 1061.737548828125, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.24521063640713692, "epoch": 0.1269516689032313, "frac_reward_zero_std": 0.5, "grad_norm": 0.2588927447795868, "learning_rate": 1e-06, "loss": 0.0305, "num_tokens": 322378605.0, "reward": 0.53125, "reward_std": 0.2354571670293808, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 745, "tools/generated_tokens": 5839.5859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1246.34375, "completions/mean_terminated_length": 1093.474365234375, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.2563879229128361, "epoch": 0.12712207382793364, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1865626573562622, "learning_rate": 1e-06, "loss": 0.0124, "num_tokens": 322788261.0, "reward": 0.5078125, "reward_std": 0.2784985899925232, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 746, "tools/generated_tokens": 4414.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1448.48828125, "completions/mean_terminated_length": 1064.1922607421875, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.2458594087511301, "epoch": 0.12729247875263594, "frac_reward_zero_std": 0.375, "grad_norm": 0.29897037148475647, "learning_rate": 1e-06, "loss": 0.009, "num_tokens": 323250962.0, "reward": 0.30859375, "reward_std": 0.22523343563079834, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 747, "tools/generated_tokens": 5672.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1400.5859375, "completions/mean_terminated_length": 1193.680419921875, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.2490228544920683, "epoch": 0.12746288367733827, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1579890102148056, "learning_rate": 1e-06, "loss": 0.0405, "num_tokens": 323704552.0, "reward": 0.4140625, "reward_std": 0.24012479186058044, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 748, "tools/generated_tokens": 5272.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1360.1171875, "completions/mean_terminated_length": 1158.6162109375, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.24075281340628862, "epoch": 0.1276332886020406, "frac_reward_zero_std": 0.375, "grad_norm": 0.1761752963066101, "learning_rate": 1e-06, "loss": 0.0354, "num_tokens": 324134294.0, "reward": 0.33984375, "reward_std": 0.2238122820854187, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 749, "tools/generated_tokens": 4928.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1267.21875, "completions/mean_terminated_length": 1105.1697998046875, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.259521072730422, "epoch": 0.12780369352674292, "frac_reward_zero_std": 0.1875, "grad_norm": 0.20703807473182678, "learning_rate": 1e-06, "loss": 0.065, "num_tokens": 324557886.0, "reward": 0.6328125, "reward_std": 0.3279016315937042, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 750, "tools/generated_tokens": 4819.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2012.0, "completions/mean_length": 1216.98046875, "completions/mean_terminated_length": 1072.12841796875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.22728418465703726, "epoch": 0.12797409845144525, "frac_reward_zero_std": 0.375, "grad_norm": 0.17022736370563507, "learning_rate": 1e-06, "loss": 0.023, "num_tokens": 324953193.0, "reward": 0.65234375, "reward_std": 0.2394643872976303, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 751, "tools/generated_tokens": 4352.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1313.6015625, "completions/mean_terminated_length": 1234.1212158203125, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.21983722131699324, "epoch": 0.12814450337614758, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15194271504878998, "learning_rate": 1e-06, "loss": 0.0064, "num_tokens": 325362499.0, "reward": 0.5078125, "reward_std": 0.15723668038845062, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 752, "tools/generated_tokens": 3585.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1363.44921875, "completions/mean_terminated_length": 1184.72900390625, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.25913594383746386, "epoch": 0.1283149083008499, "frac_reward_zero_std": 0.375, "grad_norm": 0.1837807297706604, "learning_rate": 1e-06, "loss": 0.0251, "num_tokens": 325795686.0, "reward": 0.33984375, "reward_std": 0.23348368704319, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 753, "tools/generated_tokens": 4643.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1514.44921875, "completions/mean_terminated_length": 1325.3121337890625, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.24923780746757984, "epoch": 0.1284853132255522, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1908637136220932, "learning_rate": 1e-06, "loss": 0.0119, "num_tokens": 326262649.0, "reward": 0.55859375, "reward_std": 0.21669067442417145, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 754, "tools/generated_tokens": 4418.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1263.05859375, "completions/mean_terminated_length": 1104.5963134765625, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.20850294083356857, "epoch": 0.12865571815025453, "frac_reward_zero_std": 0.5, "grad_norm": 0.17914734780788422, "learning_rate": 1e-06, "loss": -0.0136, "num_tokens": 326660904.0, "reward": 0.5625, "reward_std": 0.18706360459327698, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 755, "tools/generated_tokens": 4271.06640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1337.94140625, "completions/mean_terminated_length": 1139.14501953125, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.22479182668030262, "epoch": 0.12882612307495686, "frac_reward_zero_std": 0.6875, "grad_norm": 0.15490786731243134, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 327078537.0, "reward": 0.55859375, "reward_std": 0.1475857049226761, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 756, "tools/generated_tokens": 4457.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1411.140625, "completions/mean_terminated_length": 1100.1220703125, "completions/min_length": 358.0, "completions/min_terminated_length": 358.0, "entropy": 0.22543011792004108, "epoch": 0.1289965279996592, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1745597869157791, "learning_rate": 1e-06, "loss": 0.02, "num_tokens": 327529085.0, "reward": 0.3203125, "reward_std": 0.22623121738433838, "rewards/simpleverify_reward/mean": 0.3203125, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 757, "tools/generated_tokens": 5595.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1278.5, "completions/mean_terminated_length": 1191.512939453125, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.2280335519462824, "epoch": 0.12916693292436152, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1486106514930725, "learning_rate": 1e-06, "loss": 0.0029, "num_tokens": 327939437.0, "reward": 0.4765625, "reward_std": 0.15985959768295288, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 758, "tools/generated_tokens": 4174.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1222.5625, "completions/mean_terminated_length": 1060.5606689453125, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.241200378164649, "epoch": 0.12933733784906384, "frac_reward_zero_std": 0.25, "grad_norm": 0.21343690156936646, "learning_rate": 1e-06, "loss": 0.027, "num_tokens": 328341213.0, "reward": 0.484375, "reward_std": 0.3007515072822571, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 759, "tools/generated_tokens": 4782.55859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1250.0703125, "completions/mean_terminated_length": 1136.0848388671875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.24906747601926327, "epoch": 0.12950774277376617, "frac_reward_zero_std": 0.5, "grad_norm": 0.18144749104976654, "learning_rate": 1e-06, "loss": 0.0394, "num_tokens": 328736703.0, "reward": 0.5, "reward_std": 0.1760813295841217, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 760, "tools/generated_tokens": 4282.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1209.4140625, "completions/mean_terminated_length": 1010.9226684570312, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.21843723859637976, "epoch": 0.1296781476984685, "frac_reward_zero_std": 0.625, "grad_norm": 0.15147215127944946, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 329132265.0, "reward": 0.40625, "reward_std": 0.15364307165145874, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 761, "tools/generated_tokens": 4529.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.32421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1392.66015625, "completions/mean_terminated_length": 1078.2542724609375, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.22080306615680456, "epoch": 0.1298485526231708, "frac_reward_zero_std": 0.25, "grad_norm": 0.23827821016311646, "learning_rate": 1e-06, "loss": 0.0302, "num_tokens": 329585138.0, "reward": 0.41796875, "reward_std": 0.3341853618621826, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 762, "tools/generated_tokens": 5624.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1427.1171875, "completions/mean_terminated_length": 1207.0263671875, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.23719217535108328, "epoch": 0.13001895754787313, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20637406408786774, "learning_rate": 1e-06, "loss": 0.0076, "num_tokens": 330033824.0, "reward": 0.4609375, "reward_std": 0.24900490045547485, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 763, "tools/generated_tokens": 5299.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1232.45703125, "completions/mean_terminated_length": 1132.3026123046875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22648604400455952, "epoch": 0.13018936247257545, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18517160415649414, "learning_rate": 1e-06, "loss": -0.0135, "num_tokens": 330424517.0, "reward": 0.58203125, "reward_std": 0.246024489402771, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 764, "tools/generated_tokens": 4024.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1227.23046875, "completions/mean_terminated_length": 1126.4342041015625, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.22353226132690907, "epoch": 0.13035976739727778, "frac_reward_zero_std": 0.375, "grad_norm": 0.6939458250999451, "learning_rate": 1e-06, "loss": 0.0293, "num_tokens": 330817328.0, "reward": 0.55078125, "reward_std": 0.26575854420661926, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 765, "tools/generated_tokens": 4155.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1266.28125, "completions/mean_terminated_length": 1090.488037109375, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.21640700567513704, "epoch": 0.1305301723219801, "frac_reward_zero_std": 0.5, "grad_norm": 0.19001305103302002, "learning_rate": 1e-06, "loss": 0.0068, "num_tokens": 331223272.0, "reward": 0.5078125, "reward_std": 0.19828036427497864, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 766, "tools/generated_tokens": 4402.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1291.35546875, "completions/mean_terminated_length": 1079.4949951171875, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.24471392016857862, "epoch": 0.13070057724668244, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18988561630249023, "learning_rate": 1e-06, "loss": 0.0067, "num_tokens": 331632483.0, "reward": 0.5625, "reward_std": 0.2298790067434311, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 767, "tools/generated_tokens": 4459.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1321.8984375, "completions/mean_terminated_length": 1094.759033203125, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.22367277555167675, "epoch": 0.13087098217138476, "frac_reward_zero_std": 0.375, "grad_norm": 0.2953495383262634, "learning_rate": 1e-06, "loss": -0.0109, "num_tokens": 332052249.0, "reward": 0.54296875, "reward_std": 0.23544135689735413, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 768, "tools/generated_tokens": 4553.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1230.19921875, "completions/mean_terminated_length": 1125.731201171875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.205205911770463, "epoch": 0.13104138709608706, "frac_reward_zero_std": 0.625, "grad_norm": 0.14145499467849731, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 332430012.0, "reward": 0.59765625, "reward_std": 0.15481583774089813, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 769, "tools/generated_tokens": 3542.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.27734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1443.11328125, "completions/mean_terminated_length": 1210.9676513671875, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.17452639434486628, "epoch": 0.1312117920207894, "frac_reward_zero_std": 0.25, "grad_norm": 0.16943518817424774, "learning_rate": 1e-06, "loss": 0.0336, "num_tokens": 332878681.0, "reward": 0.5546875, "reward_std": 0.3109434247016907, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 770, "tools/generated_tokens": 5091.125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1353.765625, "completions/mean_terminated_length": 1092.49462890625, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.25280464068055153, "epoch": 0.13138219694549172, "frac_reward_zero_std": 0.4375, "grad_norm": 0.20427419245243073, "learning_rate": 1e-06, "loss": 0.0314, "num_tokens": 333315901.0, "reward": 0.4921875, "reward_std": 0.23677174746990204, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 771, "tools/generated_tokens": 5161.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1242.1015625, "completions/mean_terminated_length": 1114.4842529296875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.24298510421067476, "epoch": 0.13155260187019405, "frac_reward_zero_std": 0.5, "grad_norm": 0.2038157433271408, "learning_rate": 1e-06, "loss": -0.0289, "num_tokens": 333721079.0, "reward": 0.6015625, "reward_std": 0.20255522429943085, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 772, "tools/generated_tokens": 4458.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1292.96484375, "completions/mean_terminated_length": 1131.9384765625, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2188622960820794, "epoch": 0.13172300679489637, "frac_reward_zero_std": 0.375, "grad_norm": 0.19618146121501923, "learning_rate": 1e-06, "loss": 0.0387, "num_tokens": 334127758.0, "reward": 0.6171875, "reward_std": 0.24986931681632996, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 773, "tools/generated_tokens": 4228.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1436.421875, "completions/mean_terminated_length": 1163.4576416015625, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.22219976130872965, "epoch": 0.1318934117195987, "frac_reward_zero_std": 0.375, "grad_norm": 0.17642802000045776, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 334582298.0, "reward": 0.4375, "reward_std": 0.2617560625076294, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 774, "tools/generated_tokens": 5420.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1401.9140625, "completions/mean_terminated_length": 1260.3905029296875, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "entropy": 0.22596578113734722, "epoch": 0.13206381664430103, "frac_reward_zero_std": 0.5, "grad_norm": 0.15917572379112244, "learning_rate": 1e-06, "loss": 0.0187, "num_tokens": 335025108.0, "reward": 0.50390625, "reward_std": 0.2054290622472763, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 775, "tools/generated_tokens": 4289.91796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1409.12109375, "completions/mean_terminated_length": 1097.1104736328125, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.19567883107811213, "epoch": 0.13223422156900336, "frac_reward_zero_std": 0.3125, "grad_norm": 0.24766862392425537, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 335470179.0, "reward": 0.52734375, "reward_std": 0.27840593457221985, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 776, "tools/generated_tokens": 5441.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.96875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1274.79296875, "completions/mean_terminated_length": 1160.372314453125, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.2464032955467701, "epoch": 0.13240462649370566, "frac_reward_zero_std": 0.375, "grad_norm": 0.18497081100940704, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 335877982.0, "reward": 0.44140625, "reward_std": 0.24425500631332397, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 777, "tools/generated_tokens": 4826.796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1283.06640625, "completions/mean_terminated_length": 1192.8778076171875, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.195402885787189, "epoch": 0.13257503141840798, "frac_reward_zero_std": 0.5, "grad_norm": 0.20464791357517242, "learning_rate": 1e-06, "loss": 0.0229, "num_tokens": 336277679.0, "reward": 0.671875, "reward_std": 0.20443323254585266, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 778, "tools/generated_tokens": 3699.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1379.1796875, "completions/mean_terminated_length": 1255.3240966796875, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.2305822717025876, "epoch": 0.1327454363431103, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19441990554332733, "learning_rate": 1e-06, "loss": 0.0189, "num_tokens": 336713437.0, "reward": 0.48828125, "reward_std": 0.2589833438396454, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 779, "tools/generated_tokens": 4659.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1306.515625, "completions/mean_terminated_length": 1192.95947265625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.22024625819176435, "epoch": 0.13291584126781264, "frac_reward_zero_std": 0.1875, "grad_norm": 0.23439627885818481, "learning_rate": 1e-06, "loss": 0.0353, "num_tokens": 337128497.0, "reward": 0.5234375, "reward_std": 0.32919546961784363, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 780, "tools/generated_tokens": 4418.53125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1106.4296875, "completions/mean_terminated_length": 1083.83203125, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.22653399221599102, "epoch": 0.13308624619251497, "frac_reward_zero_std": 0.375, "grad_norm": 0.2142978012561798, "learning_rate": 1e-06, "loss": -0.0058, "num_tokens": 337494095.0, "reward": 0.61328125, "reward_std": 0.24492931365966797, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 781, "tools/generated_tokens": 3626.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1251.58984375, "completions/mean_terminated_length": 1125.4705810546875, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.2552179265767336, "epoch": 0.1332566511172173, "frac_reward_zero_std": 0.3125, "grad_norm": 0.21872830390930176, "learning_rate": 1e-06, "loss": 0.0377, "num_tokens": 337898486.0, "reward": 0.35546875, "reward_std": 0.2791505455970764, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 782, "tools/generated_tokens": 4891.60546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1298.1484375, "completions/mean_terminated_length": 1102.38427734375, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.24485708214342594, "epoch": 0.13342705604191962, "frac_reward_zero_std": 0.5, "grad_norm": 0.20661672949790955, "learning_rate": 1e-06, "loss": 0.009, "num_tokens": 338315660.0, "reward": 0.4140625, "reward_std": 0.19423659145832062, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 783, "tools/generated_tokens": 5146.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1256.671875, "completions/mean_terminated_length": 1131.3485107421875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.21234427765011787, "epoch": 0.13359746096662192, "frac_reward_zero_std": 0.5, "grad_norm": 0.19604800641536713, "learning_rate": 1e-06, "loss": 0.027, "num_tokens": 338710408.0, "reward": 0.4375, "reward_std": 0.19611266255378723, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 784, "tools/generated_tokens": 4104.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1299.63671875, "completions/mean_terminated_length": 1140.033203125, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.2084363466128707, "epoch": 0.13376786589132425, "frac_reward_zero_std": 0.25, "grad_norm": 0.1832629293203354, "learning_rate": 1e-06, "loss": 0.0361, "num_tokens": 339131691.0, "reward": 0.5078125, "reward_std": 0.2688092887401581, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 785, "tools/generated_tokens": 4411.64453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1382.03515625, "completions/mean_terminated_length": 1240.0047607421875, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.2076737228780985, "epoch": 0.13393827081602658, "frac_reward_zero_std": 0.5, "grad_norm": 0.1596766859292984, "learning_rate": 1e-06, "loss": 0.0276, "num_tokens": 339550420.0, "reward": 0.43359375, "reward_std": 0.19391977787017822, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 786, "tools/generated_tokens": 4110.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1295.9375, "completions/mean_terminated_length": 1060.677001953125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.20822795946151018, "epoch": 0.1341086757407289, "frac_reward_zero_std": 0.1875, "grad_norm": 0.24762286245822906, "learning_rate": 1e-06, "loss": 0.0399, "num_tokens": 339984612.0, "reward": 0.43359375, "reward_std": 0.31380629539489746, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 787, "tools/generated_tokens": 5247.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2006.0, "completions/mean_length": 1154.12109375, "completions/mean_terminated_length": 1102.4090576171875, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "entropy": 0.202706690877676, "epoch": 0.13427908066543123, "frac_reward_zero_std": 0.375, "grad_norm": 0.2033839225769043, "learning_rate": 1e-06, "loss": 0.0061, "num_tokens": 340354739.0, "reward": 0.73828125, "reward_std": 0.23622608184814453, "rewards/simpleverify_reward/mean": 0.73828125, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 788, "tools/generated_tokens": 3362.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1303.8359375, "completions/mean_terminated_length": 1095.4749755859375, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.23132546804845333, "epoch": 0.13444948559013356, "frac_reward_zero_std": 0.375, "grad_norm": 0.22079981863498688, "learning_rate": 1e-06, "loss": 0.0537, "num_tokens": 340768329.0, "reward": 0.36328125, "reward_std": 0.22910727560520172, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 789, "tools/generated_tokens": 4647.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1259.875, "completions/mean_terminated_length": 1068.58251953125, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.24689103197306395, "epoch": 0.1346198905148359, "frac_reward_zero_std": 0.375, "grad_norm": 0.17426487803459167, "learning_rate": 1e-06, "loss": 0.0199, "num_tokens": 341176297.0, "reward": 0.48046875, "reward_std": 0.23536449670791626, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 790, "tools/generated_tokens": 5019.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1253.33203125, "completions/mean_terminated_length": 1159.6375732421875, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.20872488245368004, "epoch": 0.13479029543953822, "frac_reward_zero_std": 0.5625, "grad_norm": 0.1594357043504715, "learning_rate": 1e-06, "loss": 0.0207, "num_tokens": 341570974.0, "reward": 0.48046875, "reward_std": 0.16124196350574493, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 791, "tools/generated_tokens": 4125.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1183.4140625, "completions/mean_terminated_length": 1110.14404296875, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.23165373411029577, "epoch": 0.13496070036424052, "frac_reward_zero_std": 0.375, "grad_norm": 0.2081676423549652, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 341949496.0, "reward": 0.56640625, "reward_std": 0.22399571537971497, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 792, "tools/generated_tokens": 3655.4140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1272.1640625, "completions/mean_terminated_length": 1227.2808837890625, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "entropy": 0.18793443776667118, "epoch": 0.13513110528894284, "frac_reward_zero_std": 0.25, "grad_norm": 0.20044372975826263, "learning_rate": 1e-06, "loss": 0.0461, "num_tokens": 342343842.0, "reward": 0.71875, "reward_std": 0.32158076763153076, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 793, "tools/generated_tokens": 3424.1640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1423.77734375, "completions/mean_terminated_length": 1290.6492919921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23818683344870806, "epoch": 0.13530151021364517, "frac_reward_zero_std": 0.375, "grad_norm": 0.19164706766605377, "learning_rate": 1e-06, "loss": -0.0089, "num_tokens": 342788905.0, "reward": 0.5625, "reward_std": 0.2836625576019287, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 794, "tools/generated_tokens": 5151.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1287.109375, "completions/mean_terminated_length": 1232.9874267578125, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.21182250510901213, "epoch": 0.1354719151383475, "frac_reward_zero_std": 0.6875, "grad_norm": 0.12743385136127472, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 343183925.0, "reward": 0.41015625, "reward_std": 0.12436182796955109, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 795, "tools/generated_tokens": 3479.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1166.4296875, "completions/mean_terminated_length": 1103.723876953125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.25628670770674944, "epoch": 0.13564232006304983, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2275211066007614, "learning_rate": 1e-06, "loss": -0.0129, "num_tokens": 343561315.0, "reward": 0.484375, "reward_std": 0.27587568759918213, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 796, "tools/generated_tokens": 3942.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1383.3046875, "completions/mean_terminated_length": 1291.7244873046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22418879810720682, "epoch": 0.13581272498775215, "frac_reward_zero_std": 0.375, "grad_norm": 0.21201610565185547, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 343987489.0, "reward": 0.51953125, "reward_std": 0.2889299988746643, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 797, "tools/generated_tokens": 4071.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1368.984375, "completions/mean_terminated_length": 1250.62841796875, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.22045026160776615, "epoch": 0.13598312991245448, "frac_reward_zero_std": 0.6875, "grad_norm": 0.1547163426876068, "learning_rate": 1e-06, "loss": 0.0029, "num_tokens": 344407901.0, "reward": 0.58203125, "reward_std": 0.13016413152217865, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 798, "tools/generated_tokens": 3993.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1183.9921875, "completions/mean_terminated_length": 1042.6136474609375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.2289585219696164, "epoch": 0.13615353483715678, "frac_reward_zero_std": 0.375, "grad_norm": 0.29640427231788635, "learning_rate": 1e-06, "loss": 0.0455, "num_tokens": 344790139.0, "reward": 0.6015625, "reward_std": 0.24570295214653015, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 799, "tools/generated_tokens": 4288.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1174.734375, "completions/mean_terminated_length": 1058.814208984375, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.2375791324302554, "epoch": 0.1363239397618591, "frac_reward_zero_std": 0.5, "grad_norm": 0.1987527310848236, "learning_rate": 1e-06, "loss": 0.0045, "num_tokens": 345168391.0, "reward": 0.421875, "reward_std": 0.16872048377990723, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 800, "tools/generated_tokens": 4086.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1137.36328125, "completions/mean_terminated_length": 1025.53076171875, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.2413704339414835, "epoch": 0.13649434468656144, "frac_reward_zero_std": 0.5625, "grad_norm": 0.17108941078186035, "learning_rate": 1e-06, "loss": 0.0137, "num_tokens": 345532452.0, "reward": 0.42578125, "reward_std": 0.15635645389556885, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 801, "tools/generated_tokens": 3617.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1142.30078125, "completions/mean_terminated_length": 1124.259033203125, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.26826963387429714, "epoch": 0.13666474961126376, "frac_reward_zero_std": 0.4375, "grad_norm": 0.23119020462036133, "learning_rate": 1e-06, "loss": 0.0163, "num_tokens": 345902289.0, "reward": 0.4609375, "reward_std": 0.23441588878631592, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 802, "tools/generated_tokens": 3694.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1271.609375, "completions/mean_terminated_length": 1106.033203125, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.19994298368692398, "epoch": 0.1368351545359661, "frac_reward_zero_std": 0.625, "grad_norm": 0.16138856112957, "learning_rate": 1e-06, "loss": 0.0144, "num_tokens": 346303165.0, "reward": 0.7734375, "reward_std": 0.1468139886856079, "rewards/simpleverify_reward/mean": 0.7734375, "rewards/simpleverify_reward/std": 0.41942715644836426, "step": 803, "tools/generated_tokens": 4143.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1276.671875, "completions/mean_terminated_length": 1221.8116455078125, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.2121621072292328, "epoch": 0.13700555946066842, "frac_reward_zero_std": 0.625, "grad_norm": 0.16794490814208984, "learning_rate": 1e-06, "loss": 0.02, "num_tokens": 346703497.0, "reward": 0.6328125, "reward_std": 0.14106407761573792, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 804, "tools/generated_tokens": 3628.66796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1265.01953125, "completions/mean_terminated_length": 1176.512939453125, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.22833317331969738, "epoch": 0.13717596438537075, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2039792388677597, "learning_rate": 1e-06, "loss": 0.0279, "num_tokens": 347108782.0, "reward": 0.5546875, "reward_std": 0.2755298614501953, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 805, "tools/generated_tokens": 4217.01953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1321.328125, "completions/mean_terminated_length": 1202.4180908203125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22586555872112513, "epoch": 0.13734636931007307, "frac_reward_zero_std": 0.375, "grad_norm": 0.20045915246009827, "learning_rate": 1e-06, "loss": 0.0296, "num_tokens": 347527506.0, "reward": 0.55859375, "reward_std": 0.25559213757514954, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 806, "tools/generated_tokens": 4657.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1219.82421875, "completions/mean_terminated_length": 1038.414306640625, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.218058155849576, "epoch": 0.13751677423477537, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15489843487739563, "learning_rate": 1e-06, "loss": -0.0017, "num_tokens": 347917845.0, "reward": 0.6015625, "reward_std": 0.16923905909061432, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 807, "tools/generated_tokens": 3987.82421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1283.1015625, "completions/mean_terminated_length": 1149.77978515625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.2296614833176136, "epoch": 0.1376871791594777, "frac_reward_zero_std": 0.5, "grad_norm": 0.15924133360385895, "learning_rate": 1e-06, "loss": 0.0161, "num_tokens": 348332815.0, "reward": 0.38671875, "reward_std": 0.20598775148391724, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 808, "tools/generated_tokens": 4475.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1181.6015625, "completions/mean_terminated_length": 1021.1574096679688, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2067298498004675, "epoch": 0.13785758408418003, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20320157706737518, "learning_rate": 1e-06, "loss": 0.0222, "num_tokens": 348744345.0, "reward": 0.54296875, "reward_std": 0.24361473321914673, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 809, "tools/generated_tokens": 4157.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1325.54296875, "completions/mean_terminated_length": 1274.15478515625, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.21559187397360802, "epoch": 0.13802798900888236, "frac_reward_zero_std": 0.5, "grad_norm": 0.18362054228782654, "learning_rate": 1e-06, "loss": 0.0258, "num_tokens": 349144196.0, "reward": 0.68359375, "reward_std": 0.19744305312633514, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 810, "tools/generated_tokens": 3229.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.26171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1277.98046875, "completions/mean_terminated_length": 1005.0211181640625, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.28414052817970514, "epoch": 0.13819839393358468, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18879370391368866, "learning_rate": 1e-06, "loss": -0.0018, "num_tokens": 349563599.0, "reward": 0.41015625, "reward_std": 0.20730705559253693, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 811, "tools/generated_tokens": 5165.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1352.12890625, "completions/mean_terminated_length": 1170.4581298828125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2402975307777524, "epoch": 0.138368798858287, "frac_reward_zero_std": 0.25, "grad_norm": 0.22159968316555023, "learning_rate": 1e-06, "loss": 0.0355, "num_tokens": 349992080.0, "reward": 0.578125, "reward_std": 0.2767269015312195, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 812, "tools/generated_tokens": 4976.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1361.875, "completions/mean_terminated_length": 1103.6666259765625, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.23502462450414896, "epoch": 0.13853920378298934, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18354931473731995, "learning_rate": 1e-06, "loss": 0.0266, "num_tokens": 350430016.0, "reward": 0.4140625, "reward_std": 0.21752606332302094, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 813, "tools/generated_tokens": 5257.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.90234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1399.40234375, "completions/mean_terminated_length": 1200.857177734375, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.23256792780011892, "epoch": 0.13870960870769164, "frac_reward_zero_std": 0.5625, "grad_norm": 0.176396906375885, "learning_rate": 1e-06, "loss": 0.0158, "num_tokens": 350867047.0, "reward": 0.44921875, "reward_std": 0.16877499222755432, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 814, "tools/generated_tokens": 4775.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1314.77734375, "completions/mean_terminated_length": 1228.3363037109375, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.22808466758579016, "epoch": 0.13888001363239397, "frac_reward_zero_std": 0.5625, "grad_norm": 0.26083889603614807, "learning_rate": 1e-06, "loss": -0.0034, "num_tokens": 351280910.0, "reward": 0.56640625, "reward_std": 0.1604563295841217, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 815, "tools/generated_tokens": 3906.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1413.96484375, "completions/mean_terminated_length": 1146.2611083984375, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.24262877833098173, "epoch": 0.1390504185570963, "frac_reward_zero_std": 0.4375, "grad_norm": 0.21163709461688995, "learning_rate": 1e-06, "loss": 0.0116, "num_tokens": 351725845.0, "reward": 0.46875, "reward_std": 0.18056906759738922, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 816, "tools/generated_tokens": 5533.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.01171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1245.390625, "completions/mean_terminated_length": 1142.8546142578125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2075537694618106, "epoch": 0.13922082348179862, "frac_reward_zero_std": 0.6875, "grad_norm": 0.1580284833908081, "learning_rate": 1e-06, "loss": 0.0059, "num_tokens": 352124329.0, "reward": 0.5859375, "reward_std": 0.10331955552101135, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 817, "tools/generated_tokens": 3693.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1213.72265625, "completions/mean_terminated_length": 1115.358154296875, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.23233703058212996, "epoch": 0.13939122840650095, "frac_reward_zero_std": 0.625, "grad_norm": 0.20453108847141266, "learning_rate": 1e-06, "loss": -0.0097, "num_tokens": 352502994.0, "reward": 0.49609375, "reward_std": 0.13699322938919067, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 818, "tools/generated_tokens": 3893.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1331.66015625, "completions/mean_terminated_length": 1153.44873046875, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.29249880835413933, "epoch": 0.13956163333120328, "frac_reward_zero_std": 0.5625, "grad_norm": 0.19395296275615692, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 352937467.0, "reward": 0.328125, "reward_std": 0.19499439001083374, "rewards/simpleverify_reward/mean": 0.328125, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 819, "tools/generated_tokens": 5219.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1098.1875, "completions/mean_terminated_length": 1013.3106079101562, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.23998859897255898, "epoch": 0.1397320382559056, "frac_reward_zero_std": 0.1875, "grad_norm": 0.2685585618019104, "learning_rate": 1e-06, "loss": 0.0363, "num_tokens": 353293723.0, "reward": 0.66015625, "reward_std": 0.29257601499557495, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 820, "tools/generated_tokens": 3514.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1251.22265625, "completions/mean_terminated_length": 1120.8408203125, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.2335311807692051, "epoch": 0.13990244318060793, "frac_reward_zero_std": 0.5, "grad_norm": 0.1694953292608261, "learning_rate": 1e-06, "loss": 0.0288, "num_tokens": 353703524.0, "reward": 0.5078125, "reward_std": 0.18627606332302094, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 821, "tools/generated_tokens": 4611.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1335.7265625, "completions/mean_terminated_length": 1127.080810546875, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.25337381288409233, "epoch": 0.14007284810531023, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19121068716049194, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 354141662.0, "reward": 0.50390625, "reward_std": 0.23945963382720947, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 822, "tools/generated_tokens": 5263.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1272.76953125, "completions/mean_terminated_length": 1111.882080078125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.20426524244248867, "epoch": 0.14024325303001256, "frac_reward_zero_std": 0.375, "grad_norm": 0.19338412582874298, "learning_rate": 1e-06, "loss": 0.0552, "num_tokens": 354543667.0, "reward": 0.6171875, "reward_std": 0.25207993388175964, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 823, "tools/generated_tokens": 4248.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1369.9140625, "completions/mean_terminated_length": 1213.4375, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.20944975595921278, "epoch": 0.1404136579547149, "frac_reward_zero_std": 0.5625, "grad_norm": 0.12369755655527115, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 354963085.0, "reward": 0.62109375, "reward_std": 0.14979633688926697, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 824, "tools/generated_tokens": 4273.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1289.65234375, "completions/mean_terminated_length": 1188.9910888671875, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.23713061213493347, "epoch": 0.14058406287941722, "frac_reward_zero_std": 0.5, "grad_norm": 0.18087173998355865, "learning_rate": 1e-06, "loss": 0.0183, "num_tokens": 355367684.0, "reward": 0.6015625, "reward_std": 0.19519630074501038, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 825, "tools/generated_tokens": 4113.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1258.69140625, "completions/mean_terminated_length": 1216.4649658203125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.22486429754644632, "epoch": 0.14075446780411954, "frac_reward_zero_std": 0.6875, "grad_norm": 0.18617857992649078, "learning_rate": 1e-06, "loss": 0.024, "num_tokens": 355758965.0, "reward": 0.625, "reward_std": 0.10331955552101135, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 826, "tools/generated_tokens": 3498.6875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1299.796875, "completions/mean_terminated_length": 1192.9107666015625, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.2561670271679759, "epoch": 0.14092487272882187, "frac_reward_zero_std": 0.375, "grad_norm": 0.1966073215007782, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 356164721.0, "reward": 0.6328125, "reward_std": 0.2532769739627838, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 827, "tools/generated_tokens": 4155.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1298.41015625, "completions/mean_terminated_length": 1167.7568359375, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.23608809150755405, "epoch": 0.1410952776535242, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14709284901618958, "learning_rate": 1e-06, "loss": 0.0127, "num_tokens": 356583898.0, "reward": 0.42578125, "reward_std": 0.17308580875396729, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 828, "tools/generated_tokens": 4490.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1208.51171875, "completions/mean_terminated_length": 1167.2335205078125, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.22018022369593382, "epoch": 0.1412656825782265, "frac_reward_zero_std": 0.625, "grad_norm": 0.16509723663330078, "learning_rate": 1e-06, "loss": 0.0157, "num_tokens": 356970621.0, "reward": 0.67578125, "reward_std": 0.13801807165145874, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 829, "tools/generated_tokens": 3456.515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1042.32421875, "completions/mean_terminated_length": 1005.6842651367188, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.28787852451205254, "epoch": 0.14143608750292883, "frac_reward_zero_std": 0.625, "grad_norm": 0.14435406029224396, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 357317776.0, "reward": 0.46484375, "reward_std": 0.1156454086303711, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 830, "tools/generated_tokens": 3426.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1331.03515625, "completions/mean_terminated_length": 1202.184326171875, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.22433694265782833, "epoch": 0.14160649242763115, "frac_reward_zero_std": 0.75, "grad_norm": 0.10016655921936035, "learning_rate": 1e-06, "loss": 0.018, "num_tokens": 357725001.0, "reward": 0.59765625, "reward_std": 0.10244406759738922, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 831, "tools/generated_tokens": 3851.04296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1155.109375, "completions/mean_terminated_length": 964.687255859375, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.2447952087968588, "epoch": 0.14177689735233348, "frac_reward_zero_std": 0.5, "grad_norm": 0.1827242076396942, "learning_rate": 1e-06, "loss": 0.0317, "num_tokens": 358103781.0, "reward": 0.59765625, "reward_std": 0.20268860459327698, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 832, "tools/generated_tokens": 4315.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1283.90625, "completions/mean_terminated_length": 1182.4779052734375, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.24892454501241446, "epoch": 0.1419473022770358, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18117961287498474, "learning_rate": 1e-06, "loss": -0.0067, "num_tokens": 358512909.0, "reward": 0.59375, "reward_std": 0.21182379126548767, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 833, "tools/generated_tokens": 4195.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.29296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1426.8515625, "completions/mean_terminated_length": 1169.480712890625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2177568394690752, "epoch": 0.14211770720173814, "frac_reward_zero_std": 0.375, "grad_norm": 0.18634583055973053, "learning_rate": 1e-06, "loss": 0.0096, "num_tokens": 358957543.0, "reward": 0.48828125, "reward_std": 0.26000863313674927, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 834, "tools/generated_tokens": 4834.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1999.0, "completions/mean_length": 1295.06640625, "completions/mean_terminated_length": 1022.7340087890625, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.23896176554262638, "epoch": 0.14228811212644046, "frac_reward_zero_std": 0.5625, "grad_norm": 0.20286424458026886, "learning_rate": 1e-06, "loss": 0.0031, "num_tokens": 359365480.0, "reward": 0.55078125, "reward_std": 0.17263562977313995, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 835, "tools/generated_tokens": 4711.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1253.16015625, "completions/mean_terminated_length": 1123.0999755859375, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.2961372844874859, "epoch": 0.1424585170511428, "frac_reward_zero_std": 0.4375, "grad_norm": 0.21413151919841766, "learning_rate": 1e-06, "loss": 0.0304, "num_tokens": 359782977.0, "reward": 0.453125, "reward_std": 0.2161029726266861, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 836, "tools/generated_tokens": 4565.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1333.03125, "completions/mean_terminated_length": 1230.8973388671875, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.245187783613801, "epoch": 0.1426289219758451, "frac_reward_zero_std": 0.6875, "grad_norm": 0.13217699527740479, "learning_rate": 1e-06, "loss": 0.0154, "num_tokens": 360192697.0, "reward": 0.8359375, "reward_std": 0.12939241528511047, "rewards/simpleverify_reward/mean": 0.8359375, "rewards/simpleverify_reward/std": 0.3710577189922333, "step": 837, "tools/generated_tokens": 3733.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1322.84765625, "completions/mean_terminated_length": 1168.1990966796875, "completions/min_length": 332.0, "completions/min_terminated_length": 332.0, "entropy": 0.2578182676807046, "epoch": 0.14279932690054742, "frac_reward_zero_std": 0.375, "grad_norm": 0.184224933385849, "learning_rate": 1e-06, "loss": 0.0204, "num_tokens": 360615954.0, "reward": 0.55078125, "reward_std": 0.2342844307422638, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 838, "tools/generated_tokens": 4594.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 1984.0, "completions/mean_length": 1177.79296875, "completions/mean_terminated_length": 1026.10546875, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.24170633126050234, "epoch": 0.14296973182524975, "frac_reward_zero_std": 0.75, "grad_norm": 0.156855970621109, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 361006941.0, "reward": 0.671875, "reward_std": 0.08351518213748932, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 839, "tools/generated_tokens": 4057.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1183.8515625, "completions/mean_terminated_length": 1051.5135498046875, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.24553822353482246, "epoch": 0.14314013674995207, "frac_reward_zero_std": 0.375, "grad_norm": 0.16294178366661072, "learning_rate": 1e-06, "loss": 0.029, "num_tokens": 361392727.0, "reward": 0.5390625, "reward_std": 0.21831360459327698, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 840, "tools/generated_tokens": 4223.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1131.57421875, "completions/mean_terminated_length": 1027.978271484375, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.24971517082303762, "epoch": 0.1433105416746544, "frac_reward_zero_std": 0.5625, "grad_norm": 0.18773284554481506, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 361757226.0, "reward": 0.61328125, "reward_std": 0.1512194126844406, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 841, "tools/generated_tokens": 3883.57421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1296.328125, "completions/mean_terminated_length": 1109.3267822265625, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.2859314167872071, "epoch": 0.14348094659935673, "frac_reward_zero_std": 0.5, "grad_norm": 0.20896050333976746, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 362177182.0, "reward": 0.48046875, "reward_std": 0.20377904176712036, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 842, "tools/generated_tokens": 4952.33203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1990.0, "completions/mean_length": 1177.43359375, "completions/mean_terminated_length": 1053.0670166015625, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.2695024525746703, "epoch": 0.14365135152405906, "frac_reward_zero_std": 0.625, "grad_norm": 0.13191479444503784, "learning_rate": 1e-06, "loss": 0.0135, "num_tokens": 362565677.0, "reward": 0.44140625, "reward_std": 0.1347845196723938, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 843, "tools/generated_tokens": 4561.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1288.15625, "completions/mean_terminated_length": 1112.8173828125, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.23288973979651928, "epoch": 0.14382175644876136, "frac_reward_zero_std": 0.1875, "grad_norm": 0.24114589393138885, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 362966485.0, "reward": 0.62890625, "reward_std": 0.35349398851394653, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 844, "tools/generated_tokens": 4320.17578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1126.171875, "completions/mean_terminated_length": 1056.4580078125, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.261497356928885, "epoch": 0.14399216137346368, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15535661578178406, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 363326817.0, "reward": 0.71484375, "reward_std": 0.17114415764808655, "rewards/simpleverify_reward/mean": 0.71484375, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 845, "tools/generated_tokens": 3550.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1303.8203125, "completions/mean_terminated_length": 1237.319091796875, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.23093167133629322, "epoch": 0.144162566298166, "frac_reward_zero_std": 0.5, "grad_norm": 0.14915740489959717, "learning_rate": 1e-06, "loss": 0.0054, "num_tokens": 363740259.0, "reward": 0.6640625, "reward_std": 0.1938907653093338, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 846, "tools/generated_tokens": 4175.82421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1357.70703125, "completions/mean_terminated_length": 1155.5050048828125, "completions/min_length": 308.0, "completions/min_terminated_length": 308.0, "entropy": 0.2513050399720669, "epoch": 0.14433297122286834, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17615199089050293, "learning_rate": 1e-06, "loss": 0.0428, "num_tokens": 364165736.0, "reward": 0.46875, "reward_std": 0.22765429317951202, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 847, "tools/generated_tokens": 4925.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1227.23046875, "completions/mean_terminated_length": 1118.283203125, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.2549811312928796, "epoch": 0.14450337614757067, "frac_reward_zero_std": 0.125, "grad_norm": 0.2387991100549698, "learning_rate": 1e-06, "loss": 0.043, "num_tokens": 364566739.0, "reward": 0.61328125, "reward_std": 0.2959836721420288, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 848, "tools/generated_tokens": 4091.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1233.2109375, "completions/mean_terminated_length": 1091.1834716796875, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.2781702149659395, "epoch": 0.144673781072273, "frac_reward_zero_std": 0.0, "grad_norm": 0.2847515344619751, "learning_rate": 1e-06, "loss": 0.0288, "num_tokens": 364969817.0, "reward": 0.5625, "reward_std": 0.36781418323516846, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 849, "tools/generated_tokens": 4641.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1326.09375, "completions/mean_terminated_length": 1155.207763671875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2481433106586337, "epoch": 0.14484418599697532, "frac_reward_zero_std": 0.25, "grad_norm": 0.20920488238334656, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 365387313.0, "reward": 0.51171875, "reward_std": 0.2757830321788788, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 850, "tools/generated_tokens": 4702.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1306.08203125, "completions/mean_terminated_length": 1160.4765625, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.29046835098415613, "epoch": 0.14501459092167765, "frac_reward_zero_std": 0.25, "grad_norm": 0.22018368542194366, "learning_rate": 1e-06, "loss": 0.0149, "num_tokens": 365806582.0, "reward": 0.54296875, "reward_std": 0.28210610151290894, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 851, "tools/generated_tokens": 4842.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1168.640625, "completions/mean_terminated_length": 1077.67236328125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.2492090780287981, "epoch": 0.14518499584637995, "frac_reward_zero_std": 0.375, "grad_norm": 0.2575002908706665, "learning_rate": 1e-06, "loss": 0.0154, "num_tokens": 366191978.0, "reward": 0.71484375, "reward_std": 0.25507354736328125, "rewards/simpleverify_reward/mean": 0.71484375, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 852, "tools/generated_tokens": 4104.64453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1159.765625, "completions/mean_terminated_length": 1112.246826171875, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.26809254195541143, "epoch": 0.14535540077108228, "frac_reward_zero_std": 0.5, "grad_norm": 0.21709080040454865, "learning_rate": 1e-06, "loss": 0.0064, "num_tokens": 366567502.0, "reward": 0.6953125, "reward_std": 0.2163851261138916, "rewards/simpleverify_reward/mean": 0.6953125, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 853, "tools/generated_tokens": 3975.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1090.95703125, "completions/mean_terminated_length": 987.3809814453125, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.25722133554518223, "epoch": 0.1455258056957846, "frac_reward_zero_std": 0.4375, "grad_norm": 0.22183115780353546, "learning_rate": 1e-06, "loss": 0.019, "num_tokens": 366930563.0, "reward": 0.44921875, "reward_std": 0.22523343563079834, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 854, "tools/generated_tokens": 4474.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1141.69140625, "completions/mean_terminated_length": 1077.2259521484375, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.22780149802565575, "epoch": 0.14569621062048693, "frac_reward_zero_std": 0.3125, "grad_norm": 0.23783008754253387, "learning_rate": 1e-06, "loss": -0.024, "num_tokens": 367313236.0, "reward": 0.37890625, "reward_std": 0.27682238817214966, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 855, "tools/generated_tokens": 4045.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1299.5078125, "completions/mean_terminated_length": 1108.7205810546875, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.2706407178193331, "epoch": 0.14586661554518926, "frac_reward_zero_std": 0.1875, "grad_norm": 0.23542420566082, "learning_rate": 1e-06, "loss": 0.0208, "num_tokens": 367741014.0, "reward": 0.45703125, "reward_std": 0.30191951990127563, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 856, "tools/generated_tokens": 5371.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.98828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1320.3828125, "completions/mean_terminated_length": 1097.64794921875, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "entropy": 0.24583891034126282, "epoch": 0.1460370204698916, "frac_reward_zero_std": 0.5, "grad_norm": 0.15358403325080872, "learning_rate": 1e-06, "loss": 0.0128, "num_tokens": 368163304.0, "reward": 0.3359375, "reward_std": 0.19486366212368011, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 857, "tools/generated_tokens": 4944.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1308.16015625, "completions/mean_terminated_length": 1175.2027587890625, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.23788707703351974, "epoch": 0.14620742539459392, "frac_reward_zero_std": 0.375, "grad_norm": 0.2616797387599945, "learning_rate": 1e-06, "loss": 0.0461, "num_tokens": 368588721.0, "reward": 0.56640625, "reward_std": 0.2728821039199829, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 858, "tools/generated_tokens": 4564.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1136.4140625, "completions/mean_terminated_length": 1037.757568359375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.24072004668414593, "epoch": 0.14637783031929621, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17764921486377716, "learning_rate": 1e-06, "loss": 0.0021, "num_tokens": 368962491.0, "reward": 0.6640625, "reward_std": 0.24900493025779724, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 859, "tools/generated_tokens": 3832.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1249.99609375, "completions/mean_terminated_length": 1115.1826171875, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.2607467984780669, "epoch": 0.14654823524399854, "frac_reward_zero_std": 0.375, "grad_norm": 0.19438406825065613, "learning_rate": 1e-06, "loss": 0.0155, "num_tokens": 369364074.0, "reward": 0.4921875, "reward_std": 0.22457927465438843, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 860, "tools/generated_tokens": 4490.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1199.83984375, "completions/mean_terminated_length": 1014.0571899414062, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.22630824986845255, "epoch": 0.14671864016870087, "frac_reward_zero_std": 0.25, "grad_norm": 0.20757856965065002, "learning_rate": 1e-06, "loss": 0.0283, "num_tokens": 369762145.0, "reward": 0.59765625, "reward_std": 0.2621135711669922, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 861, "tools/generated_tokens": 4807.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1460.46875, "completions/mean_terminated_length": 1295.9599609375, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.2073358790948987, "epoch": 0.1468890450934032, "frac_reward_zero_std": 0.4375, "grad_norm": 0.19643279910087585, "learning_rate": 1e-06, "loss": 0.0171, "num_tokens": 370208905.0, "reward": 0.5, "reward_std": 0.19718992710113525, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 862, "tools/generated_tokens": 4444.4765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1272.55859375, "completions/mean_terminated_length": 1177.3333740234375, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.23783125635236502, "epoch": 0.14705945001810553, "frac_reward_zero_std": 0.25, "grad_norm": 0.21768885850906372, "learning_rate": 1e-06, "loss": 0.0539, "num_tokens": 370611256.0, "reward": 0.65234375, "reward_std": 0.3256661295890808, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 863, "tools/generated_tokens": 4288.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1246.03125, "completions/mean_terminated_length": 1135.537841796875, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.23799911607056856, "epoch": 0.14722985494280785, "frac_reward_zero_std": 0.625, "grad_norm": 0.1385456621646881, "learning_rate": 1e-06, "loss": 0.0103, "num_tokens": 371007232.0, "reward": 0.58203125, "reward_std": 0.14656277000904083, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 864, "tools/generated_tokens": 4214.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 1998.0, "completions/mean_length": 1302.40625, "completions/mean_terminated_length": 1134.7415771484375, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.2315852651372552, "epoch": 0.14740025986751018, "frac_reward_zero_std": 0.25, "grad_norm": 0.23712506890296936, "learning_rate": 1e-06, "loss": 0.0337, "num_tokens": 371425208.0, "reward": 0.43359375, "reward_std": 0.2718029022216797, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 865, "tools/generated_tokens": 4678.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1190.7890625, "completions/mean_terminated_length": 1022.5560302734375, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "entropy": 0.23144039418548346, "epoch": 0.1475706647922125, "frac_reward_zero_std": 0.375, "grad_norm": 0.21178627014160156, "learning_rate": 1e-06, "loss": -0.0031, "num_tokens": 371816770.0, "reward": 0.390625, "reward_std": 0.25879859924316406, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 866, "tools/generated_tokens": 4334.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1192.51171875, "completions/mean_terminated_length": 1078.9556884765625, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.22740261629223824, "epoch": 0.1477410697169148, "frac_reward_zero_std": 0.3125, "grad_norm": 0.19851236045360565, "learning_rate": 1e-06, "loss": 0.0209, "num_tokens": 372204245.0, "reward": 0.75390625, "reward_std": 0.25119781494140625, "rewards/simpleverify_reward/mean": 0.75390625, "rewards/simpleverify_reward/std": 0.43157756328582764, "step": 867, "tools/generated_tokens": 4208.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1173.87109375, "completions/mean_terminated_length": 1087.583740234375, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.21943960059434175, "epoch": 0.14791147464161714, "frac_reward_zero_std": 0.625, "grad_norm": 0.12678663432598114, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 372576772.0, "reward": 0.70703125, "reward_std": 0.15920543670654297, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 868, "tools/generated_tokens": 3549.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1251.59375, "completions/mean_terminated_length": 1133.7489013671875, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.21778283175081015, "epoch": 0.14808187956631946, "frac_reward_zero_std": 0.375, "grad_norm": 0.20096375048160553, "learning_rate": 1e-06, "loss": 0.0602, "num_tokens": 372971404.0, "reward": 0.5859375, "reward_std": 0.2649868428707123, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 869, "tools/generated_tokens": 3803.60546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1204.890625, "completions/mean_terminated_length": 1062.4473876953125, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.19960143137723207, "epoch": 0.1482522844910218, "frac_reward_zero_std": 0.3125, "grad_norm": 0.22174742817878723, "learning_rate": 1e-06, "loss": -0.0078, "num_tokens": 373357904.0, "reward": 0.4375, "reward_std": 0.23392276465892792, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 870, "tools/generated_tokens": 4180.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1217.59765625, "completions/mean_terminated_length": 1135.630859375, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.21362486109137535, "epoch": 0.14842268941572412, "frac_reward_zero_std": 0.3125, "grad_norm": 0.22009976208209991, "learning_rate": 1e-06, "loss": 0.0426, "num_tokens": 373750297.0, "reward": 0.6015625, "reward_std": 0.2805894911289215, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 871, "tools/generated_tokens": 4273.60546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1085.453125, "completions/mean_terminated_length": 967.25, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.2082717027515173, "epoch": 0.14859309434042645, "frac_reward_zero_std": 0.5, "grad_norm": 0.19985249638557434, "learning_rate": 1e-06, "loss": 0.0332, "num_tokens": 374110941.0, "reward": 0.54296875, "reward_std": 0.2113366276025772, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 872, "tools/generated_tokens": 4093.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1409.828125, "completions/mean_terminated_length": 1205.88134765625, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.21090497635304928, "epoch": 0.14876349926512877, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17528687417507172, "learning_rate": 1e-06, "loss": 0.0208, "num_tokens": 374550833.0, "reward": 0.33203125, "reward_std": 0.1991586685180664, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 873, "tools/generated_tokens": 4953.83203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1250.66796875, "completions/mean_terminated_length": 1071.368408203125, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.1789833903312683, "epoch": 0.14893390418983107, "frac_reward_zero_std": 0.625, "grad_norm": 0.16894742846488953, "learning_rate": 1e-06, "loss": 0.0213, "num_tokens": 374946844.0, "reward": 0.625, "reward_std": 0.15535868704319, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 874, "tools/generated_tokens": 4274.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1344.4453125, "completions/mean_terminated_length": 1271.6680908203125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.20491211488842964, "epoch": 0.1491043091145334, "frac_reward_zero_std": 0.25, "grad_norm": 0.23800332844257355, "learning_rate": 1e-06, "loss": 0.0262, "num_tokens": 375364094.0, "reward": 0.41796875, "reward_std": 0.3040216565132141, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 875, "tools/generated_tokens": 4248.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1265.6796875, "completions/mean_terminated_length": 1107.751220703125, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.2280396344140172, "epoch": 0.14927471403923573, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2741992175579071, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 375774604.0, "reward": 0.64453125, "reward_std": 0.25387370586395264, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 876, "tools/generated_tokens": 4601.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1310.984375, "completions/mean_terminated_length": 1166.33642578125, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.18693785648792982, "epoch": 0.14944511896393806, "frac_reward_zero_std": 0.625, "grad_norm": 0.17376099526882172, "learning_rate": 1e-06, "loss": 0.0238, "num_tokens": 376182008.0, "reward": 0.53125, "reward_std": 0.15056805312633514, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 877, "tools/generated_tokens": 4078.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1252.6953125, "completions/mean_terminated_length": 1109.7650146484375, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.20417685620486736, "epoch": 0.14961552388864038, "frac_reward_zero_std": 0.3125, "grad_norm": 0.23929616808891296, "learning_rate": 1e-06, "loss": 0.0107, "num_tokens": 376577722.0, "reward": 0.640625, "reward_std": 0.2538875937461853, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 878, "tools/generated_tokens": 4180.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1343.9296875, "completions/mean_terminated_length": 1185.607666015625, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.1986571168527007, "epoch": 0.1497859288133427, "frac_reward_zero_std": 0.4375, "grad_norm": 0.17056000232696533, "learning_rate": 1e-06, "loss": 0.0141, "num_tokens": 376988552.0, "reward": 0.49609375, "reward_std": 0.21545103192329407, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 879, "tools/generated_tokens": 4335.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1264.4453125, "completions/mean_terminated_length": 1003.2604370117188, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.23919691983610392, "epoch": 0.14995633373804504, "frac_reward_zero_std": 0.4375, "grad_norm": 0.20202361047267914, "learning_rate": 1e-06, "loss": 0.0261, "num_tokens": 377401130.0, "reward": 0.5390625, "reward_std": 0.19821478426456451, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 880, "tools/generated_tokens": 4912.453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1375.703125, "completions/mean_terminated_length": 1208.45849609375, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.1874864399433136, "epoch": 0.15012673866274737, "frac_reward_zero_std": 0.5, "grad_norm": 0.19059807062149048, "learning_rate": 1e-06, "loss": 0.0098, "num_tokens": 377824862.0, "reward": 0.546875, "reward_std": 0.20356883108615875, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 881, "tools/generated_tokens": 4351.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1278.25390625, "completions/mean_terminated_length": 1067.6318359375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.16317385714501143, "epoch": 0.15029714358744967, "frac_reward_zero_std": 0.5, "grad_norm": 0.1927434504032135, "learning_rate": 1e-06, "loss": 0.0203, "num_tokens": 378230159.0, "reward": 0.5546875, "reward_std": 0.18301509320735931, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 882, "tools/generated_tokens": 4422.26171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1092.3046875, "completions/mean_terminated_length": 1015.687744140625, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.23370731435716152, "epoch": 0.150467548512152, "frac_reward_zero_std": 0.5, "grad_norm": 0.1970743089914322, "learning_rate": 1e-06, "loss": -0.0042, "num_tokens": 378586701.0, "reward": 0.66796875, "reward_std": 0.15537451207637787, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 883, "tools/generated_tokens": 3844.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1140.59375, "completions/mean_terminated_length": 1103.707275390625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2259423155337572, "epoch": 0.15063795343685432, "frac_reward_zero_std": 0.6875, "grad_norm": 0.13444504141807556, "learning_rate": 1e-06, "loss": 0.0107, "num_tokens": 378957957.0, "reward": 0.30859375, "reward_std": 0.11046826094388962, "rewards/simpleverify_reward/mean": 0.30859375, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 884, "tools/generated_tokens": 3348.6015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.30859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1404.51953125, "completions/mean_terminated_length": 1117.31640625, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "entropy": 0.20274410769343376, "epoch": 0.15080835836155665, "frac_reward_zero_std": 0.375, "grad_norm": 0.20298629999160767, "learning_rate": 1e-06, "loss": 0.008, "num_tokens": 379394634.0, "reward": 0.39453125, "reward_std": 0.27304062247276306, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 885, "tools/generated_tokens": 5220.515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.86328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1136.1015625, "completions/mean_terminated_length": 1010.4622192382812, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.20719370152801275, "epoch": 0.15097876328625898, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2245873510837555, "learning_rate": 1e-06, "loss": 0.0016, "num_tokens": 379762052.0, "reward": 0.625, "reward_std": 0.22098566591739655, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 886, "tools/generated_tokens": 3992.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1361.9453125, "completions/mean_terminated_length": 1267.4222412109375, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.19618909480050206, "epoch": 0.1511491682109613, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18611346185207367, "learning_rate": 1e-06, "loss": 0.0204, "num_tokens": 380180118.0, "reward": 0.7265625, "reward_std": 0.2028878629207611, "rewards/simpleverify_reward/mean": 0.7265625, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 887, "tools/generated_tokens": 3873.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1240.75, "completions/mean_terminated_length": 1073.2122802734375, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.2307307319715619, "epoch": 0.15131957313566363, "frac_reward_zero_std": 0.5625, "grad_norm": 0.16559883952140808, "learning_rate": 1e-06, "loss": 0.0169, "num_tokens": 380570262.0, "reward": 0.6796875, "reward_std": 0.1813678741455078, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 888, "tools/generated_tokens": 4008.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1328.7890625, "completions/mean_terminated_length": 1175.4171142578125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.17856238782405853, "epoch": 0.15148997806036593, "frac_reward_zero_std": 0.4375, "grad_norm": 0.19203080236911774, "learning_rate": 1e-06, "loss": 0.0436, "num_tokens": 380984992.0, "reward": 0.64453125, "reward_std": 0.2495906949043274, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 889, "tools/generated_tokens": 4216.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1278.50390625, "completions/mean_terminated_length": 1140.2120361328125, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.2158465851098299, "epoch": 0.15166038298506826, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18322642147541046, "learning_rate": 1e-06, "loss": 0.0167, "num_tokens": 381391249.0, "reward": 0.44921875, "reward_std": 0.24208033084869385, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 890, "tools/generated_tokens": 4310.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1294.3984375, "completions/mean_terminated_length": 1178.986572265625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.24783035833388567, "epoch": 0.1518307879097706, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2206215113401413, "learning_rate": 1e-06, "loss": 0.002, "num_tokens": 381794055.0, "reward": 0.4453125, "reward_std": 0.25406450033187866, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 891, "tools/generated_tokens": 4190.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1200.45703125, "completions/mean_terminated_length": 1048.1336669921875, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.18117011338472366, "epoch": 0.15200119283447291, "frac_reward_zero_std": 0.3125, "grad_norm": 0.22663454711437225, "learning_rate": 1e-06, "loss": 0.0287, "num_tokens": 382189932.0, "reward": 0.61328125, "reward_std": 0.2720973491668701, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 892, "tools/generated_tokens": 4168.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1351.14453125, "completions/mean_terminated_length": 1147.0201416015625, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.21165054757148027, "epoch": 0.15217159775917524, "frac_reward_zero_std": 0.375, "grad_norm": 0.20667289197444916, "learning_rate": 1e-06, "loss": 0.0094, "num_tokens": 382622081.0, "reward": 0.50390625, "reward_std": 0.2092868983745575, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 893, "tools/generated_tokens": 4871.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1301.109375, "completions/mean_terminated_length": 1146.103759765625, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.2058579958975315, "epoch": 0.15234200268387757, "frac_reward_zero_std": 0.375, "grad_norm": 0.232611745595932, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 383042045.0, "reward": 0.53125, "reward_std": 0.21808946132659912, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 894, "tools/generated_tokens": 4501.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1296.546875, "completions/mean_terminated_length": 1173.5908203125, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.19320186413824558, "epoch": 0.1525124076085799, "frac_reward_zero_std": 0.3125, "grad_norm": 0.21120324730873108, "learning_rate": 1e-06, "loss": 0.0315, "num_tokens": 383450777.0, "reward": 0.625, "reward_std": 0.25999754667282104, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 895, "tools/generated_tokens": 4264.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1341.1640625, "completions/mean_terminated_length": 1169.6068115234375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.1922745769843459, "epoch": 0.15268281253328223, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1668403595685959, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 383865779.0, "reward": 0.53125, "reward_std": 0.2108054757118225, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 896, "tools/generated_tokens": 4181.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1342.609375, "completions/mean_terminated_length": 1204.1822509765625, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.25501332245767117, "epoch": 0.15285321745798452, "frac_reward_zero_std": 0.3125, "grad_norm": 0.21012505888938904, "learning_rate": 1e-06, "loss": 0.0145, "num_tokens": 384294639.0, "reward": 0.578125, "reward_std": 0.28183847665786743, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 897, "tools/generated_tokens": 4694.625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.63671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1284.15234375, "completions/mean_terminated_length": 1084.7388916015625, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.1856649974361062, "epoch": 0.15302362238268685, "frac_reward_zero_std": 0.3125, "grad_norm": 0.18569427728652954, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 384701590.0, "reward": 0.62109375, "reward_std": 0.2376500368118286, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 898, "tools/generated_tokens": 4604.171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1340.22265625, "completions/mean_terminated_length": 1189.284423828125, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.19474520534276962, "epoch": 0.15319402730738918, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1831023097038269, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 385131087.0, "reward": 0.66015625, "reward_std": 0.19540652632713318, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 899, "tools/generated_tokens": 4396.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.27734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1402.140625, "completions/mean_terminated_length": 1154.2918701171875, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.2115109683945775, "epoch": 0.1533644322320915, "frac_reward_zero_std": 0.625, "grad_norm": 0.1206706091761589, "learning_rate": 1e-06, "loss": -0.0026, "num_tokens": 385572579.0, "reward": 0.34375, "reward_std": 0.1441391110420227, "rewards/simpleverify_reward/mean": 0.34375, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 900, "tools/generated_tokens": 5066.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1246.53125, "completions/mean_terminated_length": 1093.6976318359375, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.23318169172853231, "epoch": 0.15353483715679384, "frac_reward_zero_std": 0.625, "grad_norm": 0.15183135867118835, "learning_rate": 1e-06, "loss": 0.0076, "num_tokens": 385968043.0, "reward": 0.3671875, "reward_std": 0.12602485716342926, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 901, "tools/generated_tokens": 4358.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1248.94921875, "completions/mean_terminated_length": 1050.1658935546875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23358885757625103, "epoch": 0.15370524208149616, "frac_reward_zero_std": 0.5, "grad_norm": 0.2017795890569687, "learning_rate": 1e-06, "loss": 0.0204, "num_tokens": 386380078.0, "reward": 0.578125, "reward_std": 0.2204269915819168, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 902, "tools/generated_tokens": 4848.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1136.34375, "completions/mean_terminated_length": 925.9663696289062, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.19472294580191374, "epoch": 0.1538756470061985, "frac_reward_zero_std": 0.375, "grad_norm": 0.21478179097175598, "learning_rate": 1e-06, "loss": -0.0075, "num_tokens": 386757430.0, "reward": 0.61328125, "reward_std": 0.23600001633167267, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 903, "tools/generated_tokens": 4472.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1266.7890625, "completions/mean_terminated_length": 1126.3870849609375, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.21280538849532604, "epoch": 0.1540460519309008, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15427833795547485, "learning_rate": 1e-06, "loss": 0.006, "num_tokens": 387160640.0, "reward": 0.5859375, "reward_std": 0.16581955552101135, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 904, "tools/generated_tokens": 4266.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1161.05078125, "completions/mean_terminated_length": 1065.060546875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.21695744525641203, "epoch": 0.15421645685560312, "frac_reward_zero_std": 0.3125, "grad_norm": 0.20708735287189484, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 387528125.0, "reward": 0.48828125, "reward_std": 0.2473640739917755, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 905, "tools/generated_tokens": 3993.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1177.70703125, "completions/mean_terminated_length": 1070.8333740234375, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.17662752140313387, "epoch": 0.15438686178030545, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1443532109260559, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 387904866.0, "reward": 0.57421875, "reward_std": 0.21190981566905975, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 906, "tools/generated_tokens": 3745.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1357.25, "completions/mean_terminated_length": 1150.3958740234375, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.21647846046835184, "epoch": 0.15455726670500777, "frac_reward_zero_std": 0.4375, "grad_norm": 0.16806887090206146, "learning_rate": 1e-06, "loss": 0.0405, "num_tokens": 388331746.0, "reward": 0.35546875, "reward_std": 0.22239765524864197, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 907, "tools/generated_tokens": 4861.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1234.98828125, "completions/mean_terminated_length": 1097.630126953125, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.19143771193921566, "epoch": 0.1547276716297101, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2009657621383667, "learning_rate": 1e-06, "loss": 0.0132, "num_tokens": 388730927.0, "reward": 0.6640625, "reward_std": 0.23140643537044525, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 908, "tools/generated_tokens": 4042.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1170.640625, "completions/mean_terminated_length": 1054.181396484375, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.20012648031115532, "epoch": 0.15489807655441243, "frac_reward_zero_std": 0.25, "grad_norm": 0.23060695827007294, "learning_rate": 1e-06, "loss": 0.024, "num_tokens": 389110627.0, "reward": 0.5546875, "reward_std": 0.25087815523147583, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 909, "tools/generated_tokens": 3882.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1316.2265625, "completions/mean_terminated_length": 1188.669677734375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.21804927103221416, "epoch": 0.15506848147911476, "frac_reward_zero_std": 0.6875, "grad_norm": 0.18070954084396362, "learning_rate": 1e-06, "loss": 0.0337, "num_tokens": 389524445.0, "reward": 0.4921875, "reward_std": 0.13503573834896088, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 910, "tools/generated_tokens": 4292.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1330.4375, "completions/mean_terminated_length": 1185.5821533203125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.18107119668275118, "epoch": 0.15523888640381708, "frac_reward_zero_std": 0.625, "grad_norm": 0.20166006684303284, "learning_rate": 1e-06, "loss": 0.0078, "num_tokens": 389937549.0, "reward": 0.70703125, "reward_std": 0.140625, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 911, "tools/generated_tokens": 4026.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1256.0625, "completions/mean_terminated_length": 1049.305419921875, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.20553940907120705, "epoch": 0.15540929132851938, "frac_reward_zero_std": 0.25, "grad_norm": 0.23448815941810608, "learning_rate": 1e-06, "loss": 0.0559, "num_tokens": 390345181.0, "reward": 0.59375, "reward_std": 0.31379520893096924, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 912, "tools/generated_tokens": 4744.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1273.60546875, "completions/mean_terminated_length": 1130.2037353515625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23516938649117947, "epoch": 0.1555796962532217, "frac_reward_zero_std": 0.25, "grad_norm": 0.31176838278770447, "learning_rate": 1e-06, "loss": 0.0179, "num_tokens": 390757384.0, "reward": 0.45703125, "reward_std": 0.29775792360305786, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 913, "tools/generated_tokens": 4697.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1287.6796875, "completions/mean_terminated_length": 1064.9595947265625, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.18021881766617298, "epoch": 0.15575010117792404, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1972075253725052, "learning_rate": 1e-06, "loss": 0.0231, "num_tokens": 391178486.0, "reward": 0.4453125, "reward_std": 0.22765429317951202, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 914, "tools/generated_tokens": 4479.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1197.62890625, "completions/mean_terminated_length": 1093.2017822265625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.17587225325405598, "epoch": 0.15592050610262637, "frac_reward_zero_std": 0.375, "grad_norm": 0.2013789713382721, "learning_rate": 1e-06, "loss": 0.0405, "num_tokens": 391564215.0, "reward": 0.44140625, "reward_std": 0.23513562977313995, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 915, "tools/generated_tokens": 3749.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1212.76953125, "completions/mean_terminated_length": 1097.693359375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.20062597934156656, "epoch": 0.1560909110273287, "frac_reward_zero_std": 0.3125, "grad_norm": 0.22711578011512756, "learning_rate": 1e-06, "loss": 0.0259, "num_tokens": 391955548.0, "reward": 0.3671875, "reward_std": 0.2617396414279938, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 916, "tools/generated_tokens": 4452.76953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1143.77734375, "completions/mean_terminated_length": 1028.2642822265625, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.2103537656366825, "epoch": 0.15626131595203102, "frac_reward_zero_std": 0.25, "grad_norm": 0.3819078505039215, "learning_rate": 1e-06, "loss": 0.0275, "num_tokens": 392318771.0, "reward": 0.6015625, "reward_std": 0.26345717906951904, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 917, "tools/generated_tokens": 3743.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1201.98828125, "completions/mean_terminated_length": 1089.6903076171875, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.2054003458470106, "epoch": 0.15643172087673335, "frac_reward_zero_std": 0.375, "grad_norm": 0.21380577981472015, "learning_rate": 1e-06, "loss": -0.0105, "num_tokens": 392700240.0, "reward": 0.74609375, "reward_std": 0.24856583774089813, "rewards/simpleverify_reward/mean": 0.74609375, "rewards/simpleverify_reward/std": 0.4360972046852112, "step": 918, "tools/generated_tokens": 3634.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1192.12890625, "completions/mean_terminated_length": 1074.2088623046875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.18549406621605158, "epoch": 0.15660212580143565, "frac_reward_zero_std": 0.5, "grad_norm": 0.2368357628583908, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 393083217.0, "reward": 0.42578125, "reward_std": 0.22039085626602173, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 919, "tools/generated_tokens": 4032.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1145.1953125, "completions/mean_terminated_length": 942.1722412109375, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.2100257547572255, "epoch": 0.15677253072613798, "frac_reward_zero_std": 0.5, "grad_norm": 0.18646378815174103, "learning_rate": 1e-06, "loss": 0.016, "num_tokens": 393456195.0, "reward": 0.4921875, "reward_std": 0.20379294455051422, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 920, "tools/generated_tokens": 4545.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1257.6796875, "completions/mean_terminated_length": 1115.6497802734375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.18473036121577024, "epoch": 0.1569429356508403, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18240104615688324, "learning_rate": 1e-06, "loss": 0.0187, "num_tokens": 393849057.0, "reward": 0.59375, "reward_std": 0.200038880109787, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 921, "tools/generated_tokens": 3633.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1241.90234375, "completions/mean_terminated_length": 1126.75, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.20337208593264222, "epoch": 0.15711334057554263, "frac_reward_zero_std": 0.3125, "grad_norm": 0.244186669588089, "learning_rate": 1e-06, "loss": -0.0163, "num_tokens": 394235896.0, "reward": 0.640625, "reward_std": 0.2629890441894531, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 922, "tools/generated_tokens": 4041.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1259.125, "completions/mean_terminated_length": 1113.0369873046875, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.1947414893656969, "epoch": 0.15728374550024496, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2345479279756546, "learning_rate": 1e-06, "loss": 0.0259, "num_tokens": 394632216.0, "reward": 0.609375, "reward_std": 0.26596033573150635, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 923, "tools/generated_tokens": 4483.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1324.84765625, "completions/mean_terminated_length": 1210.3258056640625, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.16528822854161263, "epoch": 0.1574541504249473, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1458037942647934, "learning_rate": 1e-06, "loss": 0.0349, "num_tokens": 395037009.0, "reward": 0.50390625, "reward_std": 0.20013156533241272, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 924, "tools/generated_tokens": 3564.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1370.15234375, "completions/mean_terminated_length": 1251.995361328125, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.15608528349548578, "epoch": 0.15762455534964961, "frac_reward_zero_std": 0.5625, "grad_norm": 0.18368294835090637, "learning_rate": 1e-06, "loss": -0.0132, "num_tokens": 395445784.0, "reward": 0.65625, "reward_std": 0.15746080875396729, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 925, "tools/generated_tokens": 3242.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1223.41015625, "completions/mean_terminated_length": 987.2361450195312, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.19397277850657701, "epoch": 0.15779496027435194, "frac_reward_zero_std": 0.375, "grad_norm": 0.19794364273548126, "learning_rate": 1e-06, "loss": 0.0025, "num_tokens": 395846481.0, "reward": 0.640625, "reward_std": 0.21940405666828156, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 926, "tools/generated_tokens": 4551.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1133.734375, "completions/mean_terminated_length": 1021.4649047851562, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.1861046152189374, "epoch": 0.15796536519905424, "frac_reward_zero_std": 0.25, "grad_norm": 0.1933472603559494, "learning_rate": 1e-06, "loss": 0.0618, "num_tokens": 396219517.0, "reward": 0.58203125, "reward_std": 0.27970924973487854, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 927, "tools/generated_tokens": 3845.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1169.3359375, "completions/mean_terminated_length": 1052.7080078125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.21599995903670788, "epoch": 0.15813577012375657, "frac_reward_zero_std": 0.375, "grad_norm": 0.22696195542812347, "learning_rate": 1e-06, "loss": 0.0578, "num_tokens": 396604019.0, "reward": 0.42578125, "reward_std": 0.2724819481372833, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 928, "tools/generated_tokens": 4089.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.42578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1162.99609375, "completions/mean_terminated_length": 1058.650634765625, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.1994457310065627, "epoch": 0.1583061750484589, "frac_reward_zero_std": 0.1875, "grad_norm": 0.20095284283161163, "learning_rate": 1e-06, "loss": 0.0012, "num_tokens": 396991666.0, "reward": 0.73046875, "reward_std": 0.31450045108795166, "rewards/simpleverify_reward/mean": 0.73046875, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 929, "tools/generated_tokens": 4106.99609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1176.62109375, "completions/mean_terminated_length": 1082.3203125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.19552788324654102, "epoch": 0.15847657997316122, "frac_reward_zero_std": 0.375, "grad_norm": 0.25160905718803406, "learning_rate": 1e-06, "loss": 0.028, "num_tokens": 397365649.0, "reward": 0.70703125, "reward_std": 0.2381068766117096, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 930, "tools/generated_tokens": 3832.6171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1275.07421875, "completions/mean_terminated_length": 1144.4931640625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.2244763569906354, "epoch": 0.15864698489786355, "frac_reward_zero_std": 0.375, "grad_norm": 0.233364075422287, "learning_rate": 1e-06, "loss": 0.0386, "num_tokens": 397765892.0, "reward": 0.359375, "reward_std": 0.2717758119106293, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 931, "tools/generated_tokens": 4499.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1099.98828125, "completions/mean_terminated_length": 1032.5606689453125, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.19109783880412579, "epoch": 0.15881738982256588, "frac_reward_zero_std": 0.4375, "grad_norm": 0.19230061769485474, "learning_rate": 1e-06, "loss": 0.0152, "num_tokens": 398114241.0, "reward": 0.69921875, "reward_std": 0.20291273295879364, "rewards/simpleverify_reward/mean": 0.69921875, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 932, "tools/generated_tokens": 3227.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1120.76953125, "completions/mean_terminated_length": 1020.419921875, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.2288867114111781, "epoch": 0.1589877947472682, "frac_reward_zero_std": 0.6875, "grad_norm": 0.1770629733800888, "learning_rate": 1e-06, "loss": -0.0085, "num_tokens": 398498438.0, "reward": 0.5390625, "reward_std": 0.1281953752040863, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 933, "tools/generated_tokens": 3984.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1165.0625, "completions/mean_terminated_length": 1043.413330078125, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.20659015513956547, "epoch": 0.1591581996719705, "frac_reward_zero_std": 0.5, "grad_norm": 0.17342914640903473, "learning_rate": 1e-06, "loss": 0.0243, "num_tokens": 398861878.0, "reward": 0.640625, "reward_std": 0.1900683045387268, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 934, "tools/generated_tokens": 3821.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1366.2421875, "completions/mean_terminated_length": 1162.06591796875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18999553378671408, "epoch": 0.15932860459667283, "frac_reward_zero_std": 0.5625, "grad_norm": 0.17357800900936127, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 399286980.0, "reward": 0.4609375, "reward_std": 0.1857442855834961, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 935, "tools/generated_tokens": 4846.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1336.984375, "completions/mean_terminated_length": 1172.90869140625, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.1968188900500536, "epoch": 0.15949900952137516, "frac_reward_zero_std": 0.4375, "grad_norm": 0.22350461781024933, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 399703600.0, "reward": 0.6875, "reward_std": 0.19047126173973083, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 936, "tools/generated_tokens": 4032.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1305.95703125, "completions/mean_terminated_length": 1121.3560791015625, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.217020932585001, "epoch": 0.1596694144460775, "frac_reward_zero_std": 0.1875, "grad_norm": 0.24502287805080414, "learning_rate": 1e-06, "loss": -0.014, "num_tokens": 400126325.0, "reward": 0.421875, "reward_std": 0.33309003710746765, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 937, "tools/generated_tokens": 4889.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1342.07421875, "completions/mean_terminated_length": 1106.765625, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.1982845589518547, "epoch": 0.15983981937077982, "frac_reward_zero_std": 0.5625, "grad_norm": 0.187627911567688, "learning_rate": 1e-06, "loss": -0.0023, "num_tokens": 400550472.0, "reward": 0.4453125, "reward_std": 0.17278027534484863, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 938, "tools/generated_tokens": 4750.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1264.84765625, "completions/mean_terminated_length": 1164.810546875, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.24043539352715015, "epoch": 0.16001022429548215, "frac_reward_zero_std": 0.4375, "grad_norm": 0.24755193293094635, "learning_rate": 1e-06, "loss": 0.0113, "num_tokens": 400954257.0, "reward": 0.58203125, "reward_std": 0.21994972229003906, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 939, "tools/generated_tokens": 4320.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1276.05859375, "completions/mean_terminated_length": 1111.4266357421875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2400244725868106, "epoch": 0.16018062922018447, "frac_reward_zero_std": 0.625, "grad_norm": 0.24372680485248566, "learning_rate": 1e-06, "loss": 0.0201, "num_tokens": 401363920.0, "reward": 0.546875, "reward_std": 0.13896197080612183, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 940, "tools/generated_tokens": 4292.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1179.33203125, "completions/mean_terminated_length": 1032.5753173828125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.21102703362703323, "epoch": 0.1603510341448868, "frac_reward_zero_std": 0.1875, "grad_norm": 0.2658357620239258, "learning_rate": 1e-06, "loss": 0.0311, "num_tokens": 401748581.0, "reward": 0.54296875, "reward_std": 0.32371947169303894, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 941, "tools/generated_tokens": 4155.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1272.9140625, "completions/mean_terminated_length": 1112.056640625, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22449059505015612, "epoch": 0.1605214390695891, "frac_reward_zero_std": 0.5, "grad_norm": 0.19246172904968262, "learning_rate": 1e-06, "loss": 0.0102, "num_tokens": 402152543.0, "reward": 0.4453125, "reward_std": 0.20106375217437744, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 942, "tools/generated_tokens": 4584.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1186.09765625, "completions/mean_terminated_length": 1054.09912109375, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.21194147039204836, "epoch": 0.16069184399429143, "frac_reward_zero_std": 0.25, "grad_norm": 0.2552310526371002, "learning_rate": 1e-06, "loss": -0.0058, "num_tokens": 402538872.0, "reward": 0.33203125, "reward_std": 0.27799171209335327, "rewards/simpleverify_reward/mean": 0.33203125, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 943, "tools/generated_tokens": 4330.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1106.9609375, "completions/mean_terminated_length": 1040.0250244140625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.19545185193419456, "epoch": 0.16086224891899376, "frac_reward_zero_std": 0.5, "grad_norm": 0.15534307062625885, "learning_rate": 1e-06, "loss": 0.0433, "num_tokens": 402894190.0, "reward": 0.546875, "reward_std": 0.1898059844970703, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 944, "tools/generated_tokens": 3722.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1353.01171875, "completions/mean_terminated_length": 1208.7783203125, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.17680383892729878, "epoch": 0.16103265384369608, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15160922706127167, "learning_rate": 1e-06, "loss": 0.0074, "num_tokens": 403308593.0, "reward": 0.51171875, "reward_std": 0.17399311065673828, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 945, "tools/generated_tokens": 4017.0234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1167.4921875, "completions/mean_terminated_length": 1041.7054443359375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.19234557263553143, "epoch": 0.1612030587683984, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2004779428243637, "learning_rate": 1e-06, "loss": 0.0113, "num_tokens": 403684095.0, "reward": 0.6796875, "reward_std": 0.21445102989673615, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 946, "tools/generated_tokens": 3695.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1355.140625, "completions/mean_terminated_length": 1234.3760986328125, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.19640544150024652, "epoch": 0.16137346369310074, "frac_reward_zero_std": 0.375, "grad_norm": 0.1715097427368164, "learning_rate": 1e-06, "loss": 0.0064, "num_tokens": 404107219.0, "reward": 0.43359375, "reward_std": 0.2175418734550476, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 947, "tools/generated_tokens": 3915.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1080.94140625, "completions/mean_terminated_length": 994.5233764648438, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.25844348035752773, "epoch": 0.16154386861780307, "frac_reward_zero_std": 0.25, "grad_norm": 0.288810670375824, "learning_rate": 1e-06, "loss": -0.008, "num_tokens": 404473140.0, "reward": 0.48828125, "reward_std": 0.3043562173843384, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 948, "tools/generated_tokens": 4504.94140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1190.6171875, "completions/mean_terminated_length": 1076.8096923828125, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.20195814687758684, "epoch": 0.16171427354250537, "frac_reward_zero_std": 0.375, "grad_norm": 0.20876309275627136, "learning_rate": 1e-06, "loss": 0.0089, "num_tokens": 404847442.0, "reward": 0.57421875, "reward_std": 0.2607692778110504, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 949, "tools/generated_tokens": 3862.625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1257.79296875, "completions/mean_terminated_length": 1098.267578125, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.21015969943255186, "epoch": 0.1618846784672077, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2141178846359253, "learning_rate": 1e-06, "loss": 0.0167, "num_tokens": 405259853.0, "reward": 0.4140625, "reward_std": 0.24502673745155334, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 950, "tools/generated_tokens": 4609.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.63671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1153.875, "completions/mean_terminated_length": 1039.6607666015625, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.21854657400399446, "epoch": 0.16205508339191002, "frac_reward_zero_std": 0.25, "grad_norm": 0.2705812454223633, "learning_rate": 1e-06, "loss": 0.0238, "num_tokens": 405636029.0, "reward": 0.59375, "reward_std": 0.2981289029121399, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 951, "tools/generated_tokens": 3657.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1120.4921875, "completions/mean_terminated_length": 973.6018676757812, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.21976852603256702, "epoch": 0.16222548831661235, "frac_reward_zero_std": 0.3125, "grad_norm": 0.24892780184745789, "learning_rate": 1e-06, "loss": 0.0135, "num_tokens": 406003275.0, "reward": 0.59375, "reward_std": 0.2539531886577606, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 952, "tools/generated_tokens": 4160.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1041.95703125, "completions/mean_terminated_length": 903.3555908203125, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.2250966327264905, "epoch": 0.16239589324131468, "frac_reward_zero_std": 0.375, "grad_norm": 0.2442435622215271, "learning_rate": 1e-06, "loss": 0.02, "num_tokens": 406352480.0, "reward": 0.58203125, "reward_std": 0.255632221698761, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 953, "tools/generated_tokens": 4065.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1265.515625, "completions/mean_terminated_length": 1180.83544921875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.19781662989407778, "epoch": 0.162566298166017, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18892425298690796, "learning_rate": 1e-06, "loss": 0.0153, "num_tokens": 406736564.0, "reward": 0.47265625, "reward_std": 0.22665932774543762, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 954, "tools/generated_tokens": 3449.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1179.8125, "completions/mean_terminated_length": 1051.33642578125, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.2072059204801917, "epoch": 0.16273670309071933, "frac_reward_zero_std": 0.375, "grad_norm": 0.3045743405818939, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 407119156.0, "reward": 0.484375, "reward_std": 0.21433541178703308, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 955, "tools/generated_tokens": 4283.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1281.796875, "completions/mean_terminated_length": 1209.7607421875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.2013545837253332, "epoch": 0.16290710801542166, "frac_reward_zero_std": 0.5, "grad_norm": 0.2171400785446167, "learning_rate": 1e-06, "loss": -0.0003, "num_tokens": 407509248.0, "reward": 0.7421875, "reward_std": 0.16526088118553162, "rewards/simpleverify_reward/mean": 0.7421875, "rewards/simpleverify_reward/std": 0.4382871091365814, "step": 956, "tools/generated_tokens": 3489.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2008.0, "completions/mean_length": 1150.7734375, "completions/mean_terminated_length": 1013.369384765625, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.21062565967440605, "epoch": 0.16307751294012396, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2558295428752899, "learning_rate": 1e-06, "loss": 0.0383, "num_tokens": 407882486.0, "reward": 0.578125, "reward_std": 0.2771115005016327, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 957, "tools/generated_tokens": 4174.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1234.33203125, "completions/mean_terminated_length": 1172.8026123046875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.1879670936614275, "epoch": 0.1632479178648263, "frac_reward_zero_std": 0.5625, "grad_norm": 0.20825929939746857, "learning_rate": 1e-06, "loss": 0.0132, "num_tokens": 408270203.0, "reward": 0.51953125, "reward_std": 0.1550418734550476, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 958, "tools/generated_tokens": 3330.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1231.3203125, "completions/mean_terminated_length": 1146.836181640625, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.19753658398985863, "epoch": 0.16341832278952861, "frac_reward_zero_std": 0.5, "grad_norm": 0.2578915059566498, "learning_rate": 1e-06, "loss": 0.0275, "num_tokens": 408651037.0, "reward": 0.54296875, "reward_std": 0.20081061124801636, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 959, "tools/generated_tokens": 3703.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1165.65234375, "completions/mean_terminated_length": 1061.6287841796875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.1973144132643938, "epoch": 0.16358872771423094, "frac_reward_zero_std": 0.375, "grad_norm": 0.4418896436691284, "learning_rate": 1e-06, "loss": 0.0358, "num_tokens": 409025428.0, "reward": 0.76171875, "reward_std": 0.22073253989219666, "rewards/simpleverify_reward/mean": 0.76171875, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 960, "tools/generated_tokens": 3741.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1984.0, "completions/mean_length": 1351.88671875, "completions/mean_terminated_length": 1134.1334228515625, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.2260230714455247, "epoch": 0.16375913263893327, "frac_reward_zero_std": 0.375, "grad_norm": 0.3573962450027466, "learning_rate": 1e-06, "loss": 0.0469, "num_tokens": 409452327.0, "reward": 0.5, "reward_std": 0.25913122296333313, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 961, "tools/generated_tokens": 5143.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1308.61328125, "completions/mean_terminated_length": 1175.741943359375, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.20521524269133806, "epoch": 0.1639295375636356, "frac_reward_zero_std": 0.375, "grad_norm": 0.21179236471652985, "learning_rate": 1e-06, "loss": 0.0267, "num_tokens": 409858932.0, "reward": 0.5078125, "reward_std": 0.21785868704319, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 962, "tools/generated_tokens": 4276.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1232.58984375, "completions/mean_terminated_length": 1094.8310546875, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.1739743510261178, "epoch": 0.16409994248833792, "frac_reward_zero_std": 0.375, "grad_norm": 0.21300075948238373, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 410257307.0, "reward": 0.55078125, "reward_std": 0.23030208051204681, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 963, "tools/generated_tokens": 4272.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2265625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1364.80859375, "completions/mean_terminated_length": 1164.6817626953125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22997727058827877, "epoch": 0.16427034741304022, "frac_reward_zero_std": 0.375, "grad_norm": 0.21561212837696075, "learning_rate": 1e-06, "loss": 0.0133, "num_tokens": 410691706.0, "reward": 0.375, "reward_std": 0.24502672255039215, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 964, "tools/generated_tokens": 4988.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1168.1484375, "completions/mean_terminated_length": 1037.9462890625, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.1943550305441022, "epoch": 0.16444075233774255, "frac_reward_zero_std": 0.5, "grad_norm": 0.2026391625404358, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 411062496.0, "reward": 0.50390625, "reward_std": 0.209515780210495, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 965, "tools/generated_tokens": 3648.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1311.71484375, "completions/mean_terminated_length": 1124.039306640625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1801592716947198, "epoch": 0.16461115726244488, "frac_reward_zero_std": 0.4375, "grad_norm": 0.18748624622821808, "learning_rate": 1e-06, "loss": 0.018, "num_tokens": 411468439.0, "reward": 0.64453125, "reward_std": 0.23199693858623505, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 966, "tools/generated_tokens": 4183.72265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1228.61328125, "completions/mean_terminated_length": 1085.7843017578125, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.19476659875363111, "epoch": 0.1647815621871472, "frac_reward_zero_std": 0.5, "grad_norm": 0.1987626701593399, "learning_rate": 1e-06, "loss": 0.0043, "num_tokens": 411857396.0, "reward": 0.52734375, "reward_std": 0.20765095949172974, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 967, "tools/generated_tokens": 4028.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1249.9609375, "completions/mean_terminated_length": 1031.6019287109375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.19307413510978222, "epoch": 0.16495196711184953, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2244318574666977, "learning_rate": 1e-06, "loss": 0.0038, "num_tokens": 412257210.0, "reward": 0.51953125, "reward_std": 0.2428291141986847, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 968, "tools/generated_tokens": 4417.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1268.875, "completions/mean_terminated_length": 1157.575927734375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.16396487969905138, "epoch": 0.16512237203655186, "frac_reward_zero_std": 0.4375, "grad_norm": 0.1905089020729065, "learning_rate": 1e-06, "loss": 0.0368, "num_tokens": 412649162.0, "reward": 0.52734375, "reward_std": 0.20970112085342407, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 969, "tools/generated_tokens": 3532.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1328.0703125, "completions/mean_terminated_length": 1170.3857421875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.21298102103173733, "epoch": 0.1652927769612542, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2038000077009201, "learning_rate": 1e-06, "loss": 0.0095, "num_tokens": 413065244.0, "reward": 0.56640625, "reward_std": 0.21533125638961792, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 970, "tools/generated_tokens": 4264.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1234.1640625, "completions/mean_terminated_length": 1069.8779296875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.19236394576728344, "epoch": 0.16546318188595652, "frac_reward_zero_std": 0.375, "grad_norm": 0.30821314454078674, "learning_rate": 1e-06, "loss": 0.0423, "num_tokens": 413454998.0, "reward": 0.63671875, "reward_std": 0.24197597801685333, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 971, "tools/generated_tokens": 4114.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1080.46484375, "completions/mean_terminated_length": 989.5000610351562, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.18256896920502186, "epoch": 0.16563358681065882, "frac_reward_zero_std": 0.625, "grad_norm": 0.1335376352071762, "learning_rate": 1e-06, "loss": 0.0069, "num_tokens": 413805933.0, "reward": 0.6484375, "reward_std": 0.1290597915649414, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 972, "tools/generated_tokens": 3336.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1226.20703125, "completions/mean_terminated_length": 1112.9822998046875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.1840164577588439, "epoch": 0.16580399173536114, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2416093945503235, "learning_rate": 1e-06, "loss": 0.0301, "num_tokens": 414188722.0, "reward": 0.53125, "reward_std": 0.21763455867767334, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 973, "tools/generated_tokens": 3690.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1155.48828125, "completions/mean_terminated_length": 1032.5244140625, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.18110597226768732, "epoch": 0.16597439666006347, "frac_reward_zero_std": 0.4375, "grad_norm": 0.20290687680244446, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 414564111.0, "reward": 0.5234375, "reward_std": 0.19332927465438843, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 974, "tools/generated_tokens": 3979.50390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1180.51171875, "completions/mean_terminated_length": 1029.302734375, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.17978204507380724, "epoch": 0.1661448015847658, "frac_reward_zero_std": 0.5, "grad_norm": 0.28163033723831177, "learning_rate": 1e-06, "loss": 0.0058, "num_tokens": 414940434.0, "reward": 0.65234375, "reward_std": 0.18606582283973694, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 975, "tools/generated_tokens": 4044.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1200.0546875, "completions/mean_terminated_length": 1056.799072265625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.22708263341337442, "epoch": 0.16631520650946813, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2369316816329956, "learning_rate": 1e-06, "loss": 0.0155, "num_tokens": 415329808.0, "reward": 0.32421875, "reward_std": 0.14484524726867676, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 976, "tools/generated_tokens": 4008.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1239.33203125, "completions/mean_terminated_length": 1127.9244384765625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.22214957047253847, "epoch": 0.16648561143417046, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2298741340637207, "learning_rate": 1e-06, "loss": -0.0313, "num_tokens": 415739989.0, "reward": 0.32421875, "reward_std": 0.16318362951278687, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 977, "tools/generated_tokens": 4807.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1172.3515625, "completions/mean_terminated_length": 1121.6982421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.18435638770461082, "epoch": 0.16665601635887278, "frac_reward_zero_std": 0.375, "grad_norm": 0.25405094027519226, "learning_rate": 1e-06, "loss": 0.0251, "num_tokens": 416113183.0, "reward": 0.66796875, "reward_std": 0.2175418734550476, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 978, "tools/generated_tokens": 3444.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1196.91796875, "completions/mean_terminated_length": 1079.6622314453125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.19085302762687206, "epoch": 0.16682642128357508, "frac_reward_zero_std": 0.6875, "grad_norm": 0.18096967041492462, "learning_rate": 1e-06, "loss": -0.0045, "num_tokens": 416501482.0, "reward": 0.66015625, "reward_std": 0.14843884110450745, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 979, "tools/generated_tokens": 4188.9296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1293.6875, "completions/mean_terminated_length": 1047.46630859375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.20483372081071138, "epoch": 0.1669968262082774, "frac_reward_zero_std": 0.625, "grad_norm": 0.19149084389209747, "learning_rate": 1e-06, "loss": -0.0025, "num_tokens": 416925322.0, "reward": 0.4296875, "reward_std": 0.16531282663345337, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 980, "tools/generated_tokens": 4917.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1137.03125, "completions/mean_terminated_length": 1011.5244750976562, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.19713956397026777, "epoch": 0.16716723113297974, "frac_reward_zero_std": 0.25, "grad_norm": 0.2730614244937897, "learning_rate": 1e-06, "loss": 0.0402, "num_tokens": 417294562.0, "reward": 0.63671875, "reward_std": 0.29895299673080444, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 981, "tools/generated_tokens": 4281.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1198.89453125, "completions/mean_terminated_length": 1050.889892578125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.20358057040721178, "epoch": 0.16733763605768207, "frac_reward_zero_std": 0.5, "grad_norm": 0.18416868150234222, "learning_rate": 1e-06, "loss": 0.0248, "num_tokens": 417681303.0, "reward": 0.41015625, "reward_std": 0.17484626173973083, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 982, "tools/generated_tokens": 4486.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1989.0, "completions/mean_length": 1394.984375, "completions/mean_terminated_length": 1228.5343017578125, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.19067569728940725, "epoch": 0.1675080409823844, "frac_reward_zero_std": 0.625, "grad_norm": 0.1773720234632492, "learning_rate": 1e-06, "loss": 0.013, "num_tokens": 418105539.0, "reward": 0.5234375, "reward_std": 0.13862934708595276, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 983, "tools/generated_tokens": 4266.99609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1234.515625, "completions/mean_terminated_length": 1101.41357421875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.19633601233363152, "epoch": 0.16767844590708672, "frac_reward_zero_std": 0.25, "grad_norm": 0.26622453331947327, "learning_rate": 1e-06, "loss": 0.0421, "num_tokens": 418508039.0, "reward": 0.453125, "reward_std": 0.3073006868362427, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 984, "tools/generated_tokens": 4722.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1246.8125, "completions/mean_terminated_length": 1152.353759765625, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.17976173013448715, "epoch": 0.16784885083178905, "frac_reward_zero_std": 0.625, "grad_norm": 0.20381921529769897, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 418899975.0, "reward": 0.5234375, "reward_std": 0.1424899697303772, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 985, "tools/generated_tokens": 3790.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1116.67578125, "completions/mean_terminated_length": 1024.742431640625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.2113470109179616, "epoch": 0.16801925575649138, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3020055592060089, "learning_rate": 1e-06, "loss": 0.0463, "num_tokens": 419259636.0, "reward": 0.5703125, "reward_std": 0.3027361035346985, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 986, "tools/generated_tokens": 3740.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1353.3828125, "completions/mean_terminated_length": 1217.070068359375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.17643773183226585, "epoch": 0.16818966068119368, "frac_reward_zero_std": 0.25, "grad_norm": 0.278083860874176, "learning_rate": 1e-06, "loss": 0.0207, "num_tokens": 419676006.0, "reward": 0.53515625, "reward_std": 0.3173474967479706, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 987, "tools/generated_tokens": 4257.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 1987.0, "completions/mean_length": 1175.90625, "completions/mean_terminated_length": 999.8591918945312, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.17600448476150632, "epoch": 0.168360065605896, "frac_reward_zero_std": 0.4375, "grad_norm": 0.8599228262901306, "learning_rate": 1e-06, "loss": 0.0503, "num_tokens": 420054462.0, "reward": 0.68359375, "reward_std": 0.21124649047851562, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 988, "tools/generated_tokens": 3823.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1444.875, "completions/mean_terminated_length": 1239.623046875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.1784317335113883, "epoch": 0.16853047053059833, "frac_reward_zero_std": 0.6875, "grad_norm": 0.15630988776683807, "learning_rate": 1e-06, "loss": 0.0231, "num_tokens": 420491358.0, "reward": 0.46875, "reward_std": 0.13708871603012085, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 989, "tools/generated_tokens": 4388.875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1280.92578125, "completions/mean_terminated_length": 1090.1024169921875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.15464992634952068, "epoch": 0.16870087545530066, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15311479568481445, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 420898923.0, "reward": 0.484375, "reward_std": 0.16736772656440735, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 990, "tools/generated_tokens": 4176.9296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1220.37890625, "completions/mean_terminated_length": 950.2383422851562, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.20445020589977503, "epoch": 0.16887128038000299, "frac_reward_zero_std": 0.4375, "grad_norm": 0.21980640292167664, "learning_rate": 1e-06, "loss": 0.0123, "num_tokens": 421295740.0, "reward": 0.53515625, "reward_std": 0.22017385065555573, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 991, "tools/generated_tokens": 4924.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1144.33203125, "completions/mean_terminated_length": 1084.0875244140625, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.15982358064502478, "epoch": 0.1690416853047053, "frac_reward_zero_std": 0.625, "grad_norm": 0.17256851494312286, "learning_rate": 1e-06, "loss": 0.006, "num_tokens": 421657249.0, "reward": 0.59375, "reward_std": 0.15364307165145874, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 992, "tools/generated_tokens": 3064.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1210.2109375, "completions/mean_terminated_length": 1077.5294189453125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.15531280264258385, "epoch": 0.16921209022940764, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2234199047088623, "learning_rate": 1e-06, "loss": 0.0265, "num_tokens": 422055767.0, "reward": 0.51953125, "reward_std": 0.23968850076198578, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 993, "tools/generated_tokens": 4250.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1132.078125, "completions/mean_terminated_length": 1028.54345703125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.17533218674361706, "epoch": 0.16938249515410994, "frac_reward_zero_std": 0.375, "grad_norm": 0.2538447678089142, "learning_rate": 1e-06, "loss": 0.0267, "num_tokens": 422425643.0, "reward": 0.68359375, "reward_std": 0.23952803015708923, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 994, "tools/generated_tokens": 3884.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1189.30859375, "completions/mean_terminated_length": 1083.859619140625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.15691512124612927, "epoch": 0.16955290007881227, "frac_reward_zero_std": 0.4375, "grad_norm": 0.27894335985183716, "learning_rate": 1e-06, "loss": 0.0241, "num_tokens": 422801290.0, "reward": 0.51171875, "reward_std": 0.2060009390115738, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 995, "tools/generated_tokens": 3517.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1229.9921875, "completions/mean_terminated_length": 1011.3316650390625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.20765979401767254, "epoch": 0.1697233050035146, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2927990257740021, "learning_rate": 1e-06, "loss": 0.0113, "num_tokens": 423197944.0, "reward": 0.5703125, "reward_std": 0.17539192736148834, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 996, "tools/generated_tokens": 4806.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.74609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1160.9375, "completions/mean_terminated_length": 1077.53857421875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.16144005861133337, "epoch": 0.16989370992821692, "frac_reward_zero_std": 0.625, "grad_norm": 0.2151246964931488, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 423567784.0, "reward": 0.6640625, "reward_std": 0.14523236453533173, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 997, "tools/generated_tokens": 3656.94140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1187.34765625, "completions/mean_terminated_length": 1094.207763671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.17552885971963406, "epoch": 0.17006411485291925, "frac_reward_zero_std": 0.625, "grad_norm": 0.20247352123260498, "learning_rate": 1e-06, "loss": 0.0279, "num_tokens": 423946833.0, "reward": 0.6328125, "reward_std": 0.1361129879951477, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 998, "tools/generated_tokens": 3619.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1233.8515625, "completions/mean_terminated_length": 1113.3721923828125, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.18420981895178556, "epoch": 0.17023451977762158, "frac_reward_zero_std": 0.625, "grad_norm": 0.1994447112083435, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 424343691.0, "reward": 0.578125, "reward_std": 0.14711037278175354, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 999, "tools/generated_tokens": 3737.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1153.54296875, "completions/mean_terminated_length": 1061.012939453125, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.17195395100861788, "epoch": 0.1704049247023239, "frac_reward_zero_std": 0.5, "grad_norm": 0.1732359230518341, "learning_rate": 1e-06, "loss": 0.0353, "num_tokens": 424717334.0, "reward": 0.67578125, "reward_std": 0.16813471913337708, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1000, "tools/generated_tokens": 3785.55078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1174.78515625, "completions/mean_terminated_length": 1063.22900390625, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.17250444926321507, "epoch": 0.17057532962702623, "frac_reward_zero_std": 0.5, "grad_norm": 0.22788842022418976, "learning_rate": 1e-06, "loss": 0.0295, "num_tokens": 425094815.0, "reward": 0.6484375, "reward_std": 0.20789283514022827, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1001, "tools/generated_tokens": 3990.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2019.0, "completions/mean_length": 1160.69921875, "completions/mean_terminated_length": 1073.111572265625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.1925391498953104, "epoch": 0.17074573455172853, "frac_reward_zero_std": 0.625, "grad_norm": 0.16741400957107544, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 425466274.0, "reward": 0.546875, "reward_std": 0.15018154680728912, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1002, "tools/generated_tokens": 3920.7109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1184.08203125, "completions/mean_terminated_length": 1014.537353515625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.16217319201678038, "epoch": 0.17091613947643086, "frac_reward_zero_std": 0.4375, "grad_norm": 0.24424247443675995, "learning_rate": 1e-06, "loss": 0.0304, "num_tokens": 425842391.0, "reward": 0.61328125, "reward_std": 0.20365957915782928, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1003, "tools/generated_tokens": 3760.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1290.41015625, "completions/mean_terminated_length": 1208.4241943359375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.16573713533580303, "epoch": 0.1710865444011332, "frac_reward_zero_std": 0.5, "grad_norm": 0.24548400938510895, "learning_rate": 1e-06, "loss": 0.0249, "num_tokens": 426244192.0, "reward": 0.6171875, "reward_std": 0.22765710949897766, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1004, "tools/generated_tokens": 3778.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1194.328125, "completions/mean_terminated_length": 1040.9031982421875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.18604105431586504, "epoch": 0.17125694932583552, "frac_reward_zero_std": 0.4375, "grad_norm": 0.26810166239738464, "learning_rate": 1e-06, "loss": 0.0559, "num_tokens": 426624996.0, "reward": 0.5703125, "reward_std": 0.24268211424350739, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1005, "tools/generated_tokens": 4602.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1124.28125, "completions/mean_terminated_length": 997.0133666992188, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.14773790072649717, "epoch": 0.17142735425053784, "frac_reward_zero_std": 0.5625, "grad_norm": 0.15518134832382202, "learning_rate": 1e-06, "loss": 0.021, "num_tokens": 426984236.0, "reward": 0.59765625, "reward_std": 0.15936589241027832, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1006, "tools/generated_tokens": 3252.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1111.0390625, "completions/mean_terminated_length": 1000.5676879882812, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.18175256252288818, "epoch": 0.17159775917524017, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28226155042648315, "learning_rate": 1e-06, "loss": 0.0156, "num_tokens": 427346150.0, "reward": 0.51171875, "reward_std": 0.18738040328025818, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1007, "tools/generated_tokens": 3767.05078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1127.390625, "completions/mean_terminated_length": 951.8418579101562, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.17650297097861767, "epoch": 0.1717681640999425, "frac_reward_zero_std": 0.3125, "grad_norm": 0.27231544256210327, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 427719690.0, "reward": 0.69921875, "reward_std": 0.2463953047990799, "rewards/simpleverify_reward/mean": 0.69921875, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 1008, "tools/generated_tokens": 4303.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1356.06640625, "completions/mean_terminated_length": 1253.6727294921875, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.18221626617014408, "epoch": 0.1719385690246448, "frac_reward_zero_std": 0.5, "grad_norm": 0.24411547183990479, "learning_rate": 1e-06, "loss": 0.0144, "num_tokens": 428145467.0, "reward": 0.6328125, "reward_std": 0.20960843563079834, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1009, "tools/generated_tokens": 4204.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1235.953125, "completions/mean_terminated_length": 1048.5577392578125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.183076873421669, "epoch": 0.17210897394934713, "frac_reward_zero_std": 0.6875, "grad_norm": 0.20979449152946472, "learning_rate": 1e-06, "loss": 0.0121, "num_tokens": 428535871.0, "reward": 0.4296875, "reward_std": 0.11579003930091858, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1010, "tools/generated_tokens": 4499.9609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1296.82421875, "completions/mean_terminated_length": 1177.8597412109375, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.18818281218409538, "epoch": 0.17227937887404945, "frac_reward_zero_std": 0.5, "grad_norm": 0.266437292098999, "learning_rate": 1e-06, "loss": 0.0147, "num_tokens": 428942146.0, "reward": 0.5625, "reward_std": 0.22008532285690308, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1011, "tools/generated_tokens": 3984.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1181.65234375, "completions/mean_terminated_length": 976.5845336914062, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.18673587776720524, "epoch": 0.17244978379875178, "frac_reward_zero_std": 0.75, "grad_norm": 0.16682961583137512, "learning_rate": 1e-06, "loss": 0.0112, "num_tokens": 429326697.0, "reward": 0.515625, "reward_std": 0.12176600098609924, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1012, "tools/generated_tokens": 4557.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1291.94921875, "completions/mean_terminated_length": 1135.0330810546875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.15793040860444307, "epoch": 0.1726201887234541, "frac_reward_zero_std": 0.625, "grad_norm": 0.1941838413476944, "learning_rate": 1e-06, "loss": 0.0276, "num_tokens": 429727916.0, "reward": 0.58203125, "reward_std": 0.12082062661647797, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1013, "tools/generated_tokens": 3675.953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1152.8984375, "completions/mean_terminated_length": 1056.0260009765625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.18178218882530928, "epoch": 0.17279059364815644, "frac_reward_zero_std": 0.5, "grad_norm": 0.22118818759918213, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 430092354.0, "reward": 0.66796875, "reward_std": 0.2360519915819168, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1014, "tools/generated_tokens": 3768.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1119.69921875, "completions/mean_terminated_length": 1019.2337646484375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.1575491912662983, "epoch": 0.17296099857285877, "frac_reward_zero_std": 0.4375, "grad_norm": 0.26240676641464233, "learning_rate": 1e-06, "loss": 0.0368, "num_tokens": 430461749.0, "reward": 0.64453125, "reward_std": 0.2119636982679367, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1015, "tools/generated_tokens": 3871.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1120.2265625, "completions/mean_terminated_length": 1041.60595703125, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.15305958967655897, "epoch": 0.1731314034975611, "frac_reward_zero_std": 0.375, "grad_norm": 0.2847234904766083, "learning_rate": 1e-06, "loss": 0.0542, "num_tokens": 430810431.0, "reward": 0.83203125, "reward_std": 0.2556303143501282, "rewards/simpleverify_reward/mean": 0.83203125, "rewards/simpleverify_reward/std": 0.3745708465576172, "step": 1016, "tools/generated_tokens": 3184.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1169.14453125, "completions/mean_terminated_length": 986.7453002929688, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.1773261334747076, "epoch": 0.1733018084222634, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4265846610069275, "learning_rate": 1e-06, "loss": 0.0591, "num_tokens": 431186068.0, "reward": 0.66015625, "reward_std": 0.2189677655696869, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1017, "tools/generated_tokens": 4353.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1118.34765625, "completions/mean_terminated_length": 1017.7359008789062, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.19070685748010874, "epoch": 0.17347221334696572, "frac_reward_zero_std": 0.4375, "grad_norm": 0.26692995429039, "learning_rate": 1e-06, "loss": 0.0269, "num_tokens": 431552013.0, "reward": 0.56640625, "reward_std": 0.22700293362140656, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1018, "tools/generated_tokens": 4062.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1194.375, "completions/mean_terminated_length": 1110.111572265625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.17178174294531345, "epoch": 0.17364261827166805, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3452503979206085, "learning_rate": 1e-06, "loss": 0.0391, "num_tokens": 431936365.0, "reward": 0.65234375, "reward_std": 0.2902497947216034, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1019, "tools/generated_tokens": 4002.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1145.5859375, "completions/mean_terminated_length": 937.3365478515625, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.1851549595594406, "epoch": 0.17381302319637038, "frac_reward_zero_std": 0.5625, "grad_norm": 0.24081751704216003, "learning_rate": 1e-06, "loss": -0.0105, "num_tokens": 432310147.0, "reward": 0.4375, "reward_std": 0.15119513869285583, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1020, "tools/generated_tokens": 4441.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1195.5546875, "completions/mean_terminated_length": 1051.5341796875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.18198186252266169, "epoch": 0.1739834281210727, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2551412582397461, "learning_rate": 1e-06, "loss": 0.0457, "num_tokens": 432700945.0, "reward": 0.5, "reward_std": 0.23590734601020813, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1021, "tools/generated_tokens": 4355.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1124.58203125, "completions/mean_terminated_length": 997.3644409179688, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.1961059495806694, "epoch": 0.17415383304577503, "frac_reward_zero_std": 0.5625, "grad_norm": 0.22556601464748383, "learning_rate": 1e-06, "loss": 0.0385, "num_tokens": 433075782.0, "reward": 0.63671875, "reward_std": 0.18463993072509766, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1022, "tools/generated_tokens": 4340.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1297.71875, "completions/mean_terminated_length": 1137.7061767578125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.16637838073074818, "epoch": 0.17432423797047736, "frac_reward_zero_std": 0.375, "grad_norm": 0.27466294169425964, "learning_rate": 1e-06, "loss": 0.0143, "num_tokens": 433476110.0, "reward": 0.64453125, "reward_std": 0.2445330172777176, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1023, "tools/generated_tokens": 4345.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1271.79296875, "completions/mean_terminated_length": 1156.9283447265625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.1721930094063282, "epoch": 0.17449464289517966, "frac_reward_zero_std": 0.4375, "grad_norm": 0.20010972023010254, "learning_rate": 1e-06, "loss": 0.0237, "num_tokens": 433876617.0, "reward": 0.45703125, "reward_std": 0.23206253349781036, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1024, "tools/generated_tokens": 4175.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1217.08984375, "completions/mean_terminated_length": 1063.2176513671875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.17451480124145746, "epoch": 0.17466504781988199, "frac_reward_zero_std": 0.4375, "grad_norm": 0.21398308873176575, "learning_rate": 1e-06, "loss": 0.0254, "num_tokens": 434263616.0, "reward": 0.54296875, "reward_std": 0.20268860459327698, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1025, "tools/generated_tokens": 4193.1015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1275.82421875, "completions/mean_terminated_length": 1124.275634765625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.16795845329761505, "epoch": 0.1748354527445843, "frac_reward_zero_std": 0.5625, "grad_norm": 0.16005466878414154, "learning_rate": 1e-06, "loss": 0.0119, "num_tokens": 434671283.0, "reward": 0.63671875, "reward_std": 0.1438203901052475, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1026, "tools/generated_tokens": 4571.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1126.375, "completions/mean_terminated_length": 994.71435546875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.19810451474040747, "epoch": 0.17500585766928664, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28478193283081055, "learning_rate": 1e-06, "loss": 0.0141, "num_tokens": 435034419.0, "reward": 0.57421875, "reward_std": 0.184413880109787, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1027, "tools/generated_tokens": 3998.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1199.140625, "completions/mean_terminated_length": 1086.464599609375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.1970541961491108, "epoch": 0.17517626259398897, "frac_reward_zero_std": 0.4375, "grad_norm": 0.34997236728668213, "learning_rate": 1e-06, "loss": 0.03, "num_tokens": 435422199.0, "reward": 0.52734375, "reward_std": 0.24198071658611298, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1028, "tools/generated_tokens": 4543.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1039.08984375, "completions/mean_terminated_length": 944.235107421875, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.18985369056463242, "epoch": 0.1753466675186913, "frac_reward_zero_std": 0.625, "grad_norm": 0.27609509229660034, "learning_rate": 1e-06, "loss": 0.0164, "num_tokens": 435767902.0, "reward": 0.5390625, "reward_std": 0.14271603524684906, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1029, "tools/generated_tokens": 3663.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1345.23828125, "completions/mean_terminated_length": 1207.3177490234375, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.1790844751521945, "epoch": 0.17551707244339362, "frac_reward_zero_std": 0.5, "grad_norm": 0.2345418632030487, "learning_rate": 1e-06, "loss": 0.0226, "num_tokens": 436180059.0, "reward": 0.48828125, "reward_std": 0.19534093141555786, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1030, "tools/generated_tokens": 4425.2421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.50390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1335.3671875, "completions/mean_terminated_length": 1222.5068359375, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.1764806993305683, "epoch": 0.17568747736809595, "frac_reward_zero_std": 0.4375, "grad_norm": 0.21388667821884155, "learning_rate": 1e-06, "loss": -0.0099, "num_tokens": 436594233.0, "reward": 0.40625, "reward_std": 0.1896837055683136, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1031, "tools/generated_tokens": 4159.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1234.39453125, "completions/mean_terminated_length": 1134.47802734375, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.1709969500079751, "epoch": 0.17585788229279825, "frac_reward_zero_std": 0.5625, "grad_norm": 0.24780946969985962, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 436970446.0, "reward": 0.734375, "reward_std": 0.17198145389556885, "rewards/simpleverify_reward/mean": 0.734375, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 1032, "tools/generated_tokens": 3498.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1156.98046875, "completions/mean_terminated_length": 1120.76416015625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.17581572011113167, "epoch": 0.17602828721750058, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2645638585090637, "learning_rate": 1e-06, "loss": -0.021, "num_tokens": 437345033.0, "reward": 0.64453125, "reward_std": 0.2361604869365692, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1033, "tools/generated_tokens": 3644.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1104.6796875, "completions/mean_terminated_length": 1029.0548095703125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.21300080697983503, "epoch": 0.1761986921422029, "frac_reward_zero_std": 0.5625, "grad_norm": 0.9390953183174133, "learning_rate": 1e-06, "loss": -0.0163, "num_tokens": 437707559.0, "reward": 0.64453125, "reward_std": 0.15436092019081116, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1034, "tools/generated_tokens": 4064.6875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1273.140625, "completions/mean_terminated_length": 1107.8863525390625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.1554754483513534, "epoch": 0.17636909706690523, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2551327347755432, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 438099963.0, "reward": 0.5625, "reward_std": 0.20214101672172546, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1035, "tools/generated_tokens": 3793.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1276.859375, "completions/mean_terminated_length": 1207.953125, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.17798514384776354, "epoch": 0.17653950199160756, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2935298681259155, "learning_rate": 1e-06, "loss": -0.0107, "num_tokens": 438513815.0, "reward": 0.4765625, "reward_std": 0.24877604842185974, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1036, "tools/generated_tokens": 4116.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1147.81640625, "completions/mean_terminated_length": 1091.7884521484375, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.17972134612500668, "epoch": 0.1767099069163099, "frac_reward_zero_std": 0.5, "grad_norm": 0.2254374474287033, "learning_rate": 1e-06, "loss": 0.0067, "num_tokens": 438884280.0, "reward": 0.46875, "reward_std": 0.19531384110450745, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1037, "tools/generated_tokens": 4027.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1176.5859375, "completions/mean_terminated_length": 1094.658203125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.191067929379642, "epoch": 0.17688031184101222, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2188420295715332, "learning_rate": 1e-06, "loss": -0.002, "num_tokens": 439261326.0, "reward": 0.59765625, "reward_std": 0.17939911782741547, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1038, "tools/generated_tokens": 4072.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1166.4296875, "completions/mean_terminated_length": 1145.2720947265625, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.17637002188712358, "epoch": 0.17705071676571452, "frac_reward_zero_std": 0.5, "grad_norm": 0.22943973541259766, "learning_rate": 1e-06, "loss": -0.0185, "num_tokens": 439624908.0, "reward": 0.72265625, "reward_std": 0.17131631076335907, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1039, "tools/generated_tokens": 3270.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.02734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1275.4375, "completions/mean_terminated_length": 1044.071044921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22527245059609413, "epoch": 0.17722112169041684, "frac_reward_zero_std": 0.4375, "grad_norm": 0.25720152258872986, "learning_rate": 1e-06, "loss": 0.0395, "num_tokens": 440040716.0, "reward": 0.359375, "reward_std": 0.20294174551963806, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1040, "tools/generated_tokens": 5667.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1214.48828125, "completions/mean_terminated_length": 1022.144287109375, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.1827351739630103, "epoch": 0.17739152661511917, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3032543361186981, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 440440329.0, "reward": 0.36328125, "reward_std": 0.21953772008419037, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1041, "tools/generated_tokens": 5358.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1334.4921875, "completions/mean_terminated_length": 1210.123779296875, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.1883723959326744, "epoch": 0.1775619315398215, "frac_reward_zero_std": 0.4375, "grad_norm": 0.22649165987968445, "learning_rate": 1e-06, "loss": 0.0348, "num_tokens": 440857991.0, "reward": 0.578125, "reward_std": 0.19036275148391724, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1042, "tools/generated_tokens": 4558.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1222.26953125, "completions/mean_terminated_length": 1087.1500244140625, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.19280875474214554, "epoch": 0.17773233646452383, "frac_reward_zero_std": 0.3125, "grad_norm": 0.33298274874687195, "learning_rate": 1e-06, "loss": 0.0368, "num_tokens": 441255820.0, "reward": 0.58203125, "reward_std": 0.28695064783096313, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1043, "tools/generated_tokens": 4702.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1060.42578125, "completions/mean_terminated_length": 994.5875244140625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.1948423534631729, "epoch": 0.17790274138922615, "frac_reward_zero_std": 0.5, "grad_norm": 0.22411775588989258, "learning_rate": 1e-06, "loss": 0.0135, "num_tokens": 441604601.0, "reward": 0.37890625, "reward_std": 0.1902293711900711, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 1044, "tools/generated_tokens": 3596.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1318.1015625, "completions/mean_terminated_length": 1174.8597412109375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.15403541550040245, "epoch": 0.17807314631392848, "frac_reward_zero_std": 0.375, "grad_norm": 0.22829307615756989, "learning_rate": 1e-06, "loss": 0.0308, "num_tokens": 442014803.0, "reward": 0.52734375, "reward_std": 0.23041057586669922, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1045, "tools/generated_tokens": 4494.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1144.0859375, "completions/mean_terminated_length": 1010.3319091796875, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.17259704135358334, "epoch": 0.1782435512386308, "frac_reward_zero_std": 0.6875, "grad_norm": 0.18448218703269958, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 442387049.0, "reward": 0.421875, "reward_std": 0.12576062977313995, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1046, "tools/generated_tokens": 4104.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1099.49609375, "completions/mean_terminated_length": 1044.6239013671875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.1896209018304944, "epoch": 0.1784139561633331, "frac_reward_zero_std": 0.3125, "grad_norm": 0.30165600776672363, "learning_rate": 1e-06, "loss": 0.0239, "num_tokens": 442747224.0, "reward": 0.625, "reward_std": 0.2546864151954651, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1047, "tools/generated_tokens": 3659.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1167.6171875, "completions/mean_terminated_length": 999.730224609375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.19375190697610378, "epoch": 0.17858436108803544, "frac_reward_zero_std": 0.4375, "grad_norm": 0.31340327858924866, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 443126886.0, "reward": 0.55078125, "reward_std": 0.20576362311840057, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1048, "tools/generated_tokens": 4695.6171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1268.85546875, "completions/mean_terminated_length": 1165.42919921875, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.18054699152708054, "epoch": 0.17875476601273776, "frac_reward_zero_std": 0.5625, "grad_norm": 0.20809531211853027, "learning_rate": 1e-06, "loss": 0.0025, "num_tokens": 443524449.0, "reward": 0.4296875, "reward_std": 0.16047024726867676, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1049, "tools/generated_tokens": 4124.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1289.57421875, "completions/mean_terminated_length": 1185.0888671875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.14327248698100448, "epoch": 0.1789251709374401, "frac_reward_zero_std": 0.5625, "grad_norm": 0.27817532420158386, "learning_rate": 1e-06, "loss": 0.011, "num_tokens": 443920708.0, "reward": 0.57421875, "reward_std": 0.13699322938919067, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1050, "tools/generated_tokens": 3881.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 1129.3828125, "completions/mean_terminated_length": 1007.4424438476562, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.1989704305306077, "epoch": 0.17909557586214242, "frac_reward_zero_std": 0.5, "grad_norm": 0.4978354871273041, "learning_rate": 1e-06, "loss": 0.0354, "num_tokens": 444299942.0, "reward": 0.3046875, "reward_std": 0.21056963503360748, "rewards/simpleverify_reward/mean": 0.3046875, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 1051, "tools/generated_tokens": 4409.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1134.7109375, "completions/mean_terminated_length": 1022.5526123046875, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.1862142002210021, "epoch": 0.17926598078684475, "frac_reward_zero_std": 0.4375, "grad_norm": 0.25547003746032715, "learning_rate": 1e-06, "loss": 0.0302, "num_tokens": 444673212.0, "reward": 0.60546875, "reward_std": 0.24425500631332397, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1052, "tools/generated_tokens": 3918.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1069.1640625, "completions/mean_terminated_length": 953.7554931640625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.19902301765978336, "epoch": 0.17943638571154708, "frac_reward_zero_std": 0.3125, "grad_norm": 0.22894960641860962, "learning_rate": 1e-06, "loss": 0.007, "num_tokens": 445025542.0, "reward": 0.41015625, "reward_std": 0.24932172894477844, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1053, "tools/generated_tokens": 4117.16015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1072.53125, "completions/mean_terminated_length": 938.1333618164062, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.22597424685955048, "epoch": 0.17960679063624937, "frac_reward_zero_std": 0.625, "grad_norm": 0.24048912525177002, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 445395150.0, "reward": 0.37890625, "reward_std": 0.17218545079231262, "rewards/simpleverify_reward/mean": 0.37890625, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 1054, "tools/generated_tokens": 4992.53125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1230.5859375, "completions/mean_terminated_length": 1126.1585693359375, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.16794088389724493, "epoch": 0.1797771955609517, "frac_reward_zero_std": 0.6875, "grad_norm": 0.1720690280199051, "learning_rate": 1e-06, "loss": 0.0136, "num_tokens": 445779044.0, "reward": 0.55859375, "reward_std": 0.12333697080612183, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1055, "tools/generated_tokens": 3478.578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1247.2109375, "completions/mean_terminated_length": 1152.7947998046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.1659185467287898, "epoch": 0.17994760048565403, "frac_reward_zero_std": 0.5, "grad_norm": 0.19820857048034668, "learning_rate": 1e-06, "loss": 0.0182, "num_tokens": 446172602.0, "reward": 0.43359375, "reward_std": 0.1750703752040863, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1056, "tools/generated_tokens": 3887.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1205.27734375, "completions/mean_terminated_length": 1105.9169921875, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.1656830022111535, "epoch": 0.18011800541035636, "frac_reward_zero_std": 0.5, "grad_norm": 0.2748764157295227, "learning_rate": 1e-06, "loss": 0.0192, "num_tokens": 446564385.0, "reward": 0.50390625, "reward_std": 0.17548459768295288, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1057, "tools/generated_tokens": 4349.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1106.578125, "completions/mean_terminated_length": 1087.82470703125, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.15531493723392487, "epoch": 0.18028841033505869, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2593691647052765, "learning_rate": 1e-06, "loss": 0.0162, "num_tokens": 446913173.0, "reward": 0.53515625, "reward_std": 0.2396475076675415, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1058, "tools/generated_tokens": 3082.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1195.7421875, "completions/mean_terminated_length": 1107.57763671875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.14528470300137997, "epoch": 0.180458815259761, "frac_reward_zero_std": 0.3125, "grad_norm": 0.24477677047252655, "learning_rate": 1e-06, "loss": 0.0061, "num_tokens": 447304051.0, "reward": 0.6484375, "reward_std": 0.2635255455970764, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1059, "tools/generated_tokens": 4235.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1217.35546875, "completions/mean_terminated_length": 1090.1396484375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.20077259466052055, "epoch": 0.18062922018446334, "frac_reward_zero_std": 0.5, "grad_norm": 0.28027454018592834, "learning_rate": 1e-06, "loss": 0.0545, "num_tokens": 447697870.0, "reward": 0.47265625, "reward_std": 0.22153326869010925, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1060, "tools/generated_tokens": 5177.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1069.11328125, "completions/mean_terminated_length": 972.4849853515625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.19247326627373695, "epoch": 0.18079962510916567, "frac_reward_zero_std": 0.5625, "grad_norm": 0.22977644205093384, "learning_rate": 1e-06, "loss": 0.0322, "num_tokens": 448048379.0, "reward": 0.32421875, "reward_std": 0.1512194126844406, "rewards/simpleverify_reward/mean": 0.32421875, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1061, "tools/generated_tokens": 4173.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1073.078125, "completions/mean_terminated_length": 938.760009765625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.18444763589650393, "epoch": 0.18097003003386797, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2830033004283905, "learning_rate": 1e-06, "loss": 0.0544, "num_tokens": 448409087.0, "reward": 0.4609375, "reward_std": 0.27752572298049927, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1062, "tools/generated_tokens": 4481.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1277.48828125, "completions/mean_terminated_length": 1139.0091552734375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.18034209497272968, "epoch": 0.1811404349585703, "frac_reward_zero_std": 0.625, "grad_norm": 0.2301025688648224, "learning_rate": 1e-06, "loss": 0.0293, "num_tokens": 448814684.0, "reward": 0.68359375, "reward_std": 0.158915713429451, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1063, "tools/generated_tokens": 4229.484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1119.66015625, "completions/mean_terminated_length": 1036.7021484375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.15907876566052437, "epoch": 0.18131083988327262, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2984767258167267, "learning_rate": 1e-06, "loss": 0.0265, "num_tokens": 449172485.0, "reward": 0.6171875, "reward_std": 0.2734277546405792, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1064, "tools/generated_tokens": 3591.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1077.9140625, "completions/mean_terminated_length": 991.2255249023438, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.1689898082986474, "epoch": 0.18148124480797495, "frac_reward_zero_std": 0.5, "grad_norm": 0.2286832630634308, "learning_rate": 1e-06, "loss": 0.0096, "num_tokens": 449522159.0, "reward": 0.5546875, "reward_std": 0.2048833966255188, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1065, "tools/generated_tokens": 3973.9609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1262.57421875, "completions/mean_terminated_length": 1125.669677734375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.20837523601949215, "epoch": 0.18165164973267728, "frac_reward_zero_std": 0.5, "grad_norm": 0.25576871633529663, "learning_rate": 1e-06, "loss": 0.043, "num_tokens": 449927410.0, "reward": 0.52734375, "reward_std": 0.2029259204864502, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1066, "tools/generated_tokens": 4694.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.67578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1208.11328125, "completions/mean_terminated_length": 1083.8251953125, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.20671589486300945, "epoch": 0.1818220546573796, "frac_reward_zero_std": 0.5, "grad_norm": 0.2632042169570923, "learning_rate": 1e-06, "loss": -0.0018, "num_tokens": 450315199.0, "reward": 0.515625, "reward_std": 0.1731128990650177, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1067, "tools/generated_tokens": 4440.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1326.66015625, "completions/mean_terminated_length": 1133.82666015625, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.16647742595523596, "epoch": 0.18199245958208193, "frac_reward_zero_std": 0.6875, "grad_norm": 0.16425293684005737, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 450724888.0, "reward": 0.5859375, "reward_std": 0.13149452209472656, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1068, "tools/generated_tokens": 4070.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1145.625, "completions/mean_terminated_length": 1002.7285766601562, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.18091893941164017, "epoch": 0.18216286450678423, "frac_reward_zero_std": 0.5, "grad_norm": 0.24490150809288025, "learning_rate": 1e-06, "loss": 0.0609, "num_tokens": 451102824.0, "reward": 0.59375, "reward_std": 0.21753078699111938, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1069, "tools/generated_tokens": 4361.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1166.09375, "completions/mean_terminated_length": 1070.6492919921875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.19607526902109385, "epoch": 0.18233326943148656, "frac_reward_zero_std": 0.625, "grad_norm": 0.23965923488140106, "learning_rate": 1e-06, "loss": 0.0004, "num_tokens": 451483680.0, "reward": 0.40625, "reward_std": 0.1399868279695511, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1070, "tools/generated_tokens": 4422.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1202.8515625, "completions/mean_terminated_length": 1127.32763671875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.20735706761479378, "epoch": 0.1825036743561889, "frac_reward_zero_std": 0.4375, "grad_norm": 0.25708523392677307, "learning_rate": 1e-06, "loss": 0.0244, "num_tokens": 451869402.0, "reward": 0.515625, "reward_std": 0.21973475813865662, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1071, "tools/generated_tokens": 4018.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1243.375, "completions/mean_terminated_length": 1071.7725830078125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.2050136774778366, "epoch": 0.18267407928089122, "frac_reward_zero_std": 0.625, "grad_norm": 0.18401972949504852, "learning_rate": 1e-06, "loss": -0.0046, "num_tokens": 452272698.0, "reward": 0.33984375, "reward_std": 0.18191072344779968, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1072, "tools/generated_tokens": 5227.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1174.515625, "completions/mean_terminated_length": 1012.763916015625, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.17206366453319788, "epoch": 0.18284448420559354, "frac_reward_zero_std": 0.3125, "grad_norm": 0.28359100222587585, "learning_rate": 1e-06, "loss": 0.0637, "num_tokens": 452646334.0, "reward": 0.5859375, "reward_std": 0.26385819911956787, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1073, "tools/generated_tokens": 4382.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2000.0, "completions/mean_length": 1085.68359375, "completions/mean_terminated_length": 1008.5400390625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.17506478633731604, "epoch": 0.18301488913029587, "frac_reward_zero_std": 0.375, "grad_norm": 0.26863619685173035, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 453012557.0, "reward": 0.46875, "reward_std": 0.2507653832435608, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1074, "tools/generated_tokens": 4173.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1056.515625, "completions/mean_terminated_length": 944.4347534179688, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.1767062684521079, "epoch": 0.1831852940549982, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3461708724498749, "learning_rate": 1e-06, "loss": 0.0059, "num_tokens": 453362513.0, "reward": 0.55078125, "reward_std": 0.2531842887401581, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1075, "tools/generated_tokens": 3864.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1132.3515625, "completions/mean_terminated_length": 957.739501953125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.19119540695101023, "epoch": 0.18335569897970053, "frac_reward_zero_std": 0.5, "grad_norm": 0.25844788551330566, "learning_rate": 1e-06, "loss": 0.0148, "num_tokens": 453737163.0, "reward": 0.4296875, "reward_std": 0.2122168391942978, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1076, "tools/generated_tokens": 4772.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1191.65234375, "completions/mean_terminated_length": 1042.38525390625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.18736120127141476, "epoch": 0.18352610390440283, "frac_reward_zero_std": 0.4375, "grad_norm": 0.49937495589256287, "learning_rate": 1e-06, "loss": 0.003, "num_tokens": 454117714.0, "reward": 0.359375, "reward_std": 0.2012898027896881, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1077, "tools/generated_tokens": 4375.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1167.11328125, "completions/mean_terminated_length": 1054.5902099609375, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.16093464195728302, "epoch": 0.18369650882910515, "frac_reward_zero_std": 0.5, "grad_norm": 0.2516961097717285, "learning_rate": 1e-06, "loss": 0.0186, "num_tokens": 454491439.0, "reward": 0.6484375, "reward_std": 0.20403026044368744, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1078, "tools/generated_tokens": 3879.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1123.14453125, "completions/mean_terminated_length": 1044.771240234375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.1910272864624858, "epoch": 0.18386691375380748, "frac_reward_zero_std": 0.5, "grad_norm": 0.239535853266716, "learning_rate": 1e-06, "loss": 0.0111, "num_tokens": 454855140.0, "reward": 0.52734375, "reward_std": 0.19683241844177246, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1079, "tools/generated_tokens": 3891.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1131.7578125, "completions/mean_terminated_length": 1019.2412109375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.16171892266720533, "epoch": 0.1840373186785098, "frac_reward_zero_std": 0.4375, "grad_norm": 0.25692424178123474, "learning_rate": 1e-06, "loss": -0.0357, "num_tokens": 455218534.0, "reward": 0.48046875, "reward_std": 0.213389590382576, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1080, "tools/generated_tokens": 4227.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1048.68359375, "completions/mean_terminated_length": 916.0309448242188, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.16091797221451998, "epoch": 0.18420772360321214, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4236818850040436, "learning_rate": 1e-06, "loss": 0.0236, "num_tokens": 455566277.0, "reward": 0.61328125, "reward_std": 0.250667929649353, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1081, "tools/generated_tokens": 4064.69140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1123.0078125, "completions/mean_terminated_length": 986.1345825195312, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.17835950572043657, "epoch": 0.18437812852791446, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2822694182395935, "learning_rate": 1e-06, "loss": 0.0117, "num_tokens": 455934135.0, "reward": 0.57421875, "reward_std": 0.2142380028963089, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1082, "tools/generated_tokens": 4307.015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1169.3671875, "completions/mean_terminated_length": 1039.3453369140625, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.1422855188138783, "epoch": 0.1845485334526168, "frac_reward_zero_std": 0.5625, "grad_norm": 0.19664648175239563, "learning_rate": 1e-06, "loss": 0.0355, "num_tokens": 456316165.0, "reward": 0.515625, "reward_std": 0.16703036427497864, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1083, "tools/generated_tokens": 4017.3671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1176.1953125, "completions/mean_terminated_length": 1069.131591796875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.1832456048578024, "epoch": 0.1847189383773191, "frac_reward_zero_std": 0.4375, "grad_norm": 0.29349762201309204, "learning_rate": 1e-06, "loss": 0.0319, "num_tokens": 456697623.0, "reward": 0.6015625, "reward_std": 0.24063712358474731, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1084, "tools/generated_tokens": 4408.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1008.19140625, "completions/mean_terminated_length": 978.9597778320312, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.16325358115136623, "epoch": 0.18488934330202142, "frac_reward_zero_std": 0.4375, "grad_norm": 0.27482786774635315, "learning_rate": 1e-06, "loss": 0.0004, "num_tokens": 457030536.0, "reward": 0.72265625, "reward_std": 0.22665932774543762, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1085, "tools/generated_tokens": 3296.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1185.203125, "completions/mean_terminated_length": 1011.0234985351562, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.17100472003221512, "epoch": 0.18505974822672375, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2283105105161667, "learning_rate": 1e-06, "loss": 0.007, "num_tokens": 457421580.0, "reward": 0.54296875, "reward_std": 0.11586952954530716, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1086, "tools/generated_tokens": 4353.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1113.68359375, "completions/mean_terminated_length": 1021.4549560546875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.15519069973379374, "epoch": 0.18523015315142607, "frac_reward_zero_std": 0.625, "grad_norm": 0.48820433020591736, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 457780411.0, "reward": 0.484375, "reward_std": 0.15678457915782928, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1087, "tools/generated_tokens": 3513.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1113.6796875, "completions/mean_terminated_length": 1071.7305908203125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.1771161425858736, "epoch": 0.1854005580761284, "frac_reward_zero_std": 0.4375, "grad_norm": 0.30359533429145813, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 458135081.0, "reward": 0.5625, "reward_std": 0.19508779048919678, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1088, "tools/generated_tokens": 3425.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1217.79296875, "completions/mean_terminated_length": 1099.1920166015625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.18789450265467167, "epoch": 0.18557096300083073, "frac_reward_zero_std": 0.4375, "grad_norm": 0.31035393476486206, "learning_rate": 1e-06, "loss": 0.0199, "num_tokens": 458520852.0, "reward": 0.51171875, "reward_std": 0.19750864803791046, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1089, "tools/generated_tokens": 4217.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1129.5, "completions/mean_terminated_length": 1111.2032470703125, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.13960112864151597, "epoch": 0.18574136792553306, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2488587200641632, "learning_rate": 1e-06, "loss": 0.0047, "num_tokens": 458876116.0, "reward": 0.76171875, "reward_std": 0.14029237627983093, "rewards/simpleverify_reward/mean": 0.76171875, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 1090, "tools/generated_tokens": 3105.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1033.87109375, "completions/mean_terminated_length": 947.927978515625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.17752600088715553, "epoch": 0.18591177285023539, "frac_reward_zero_std": 0.75, "grad_norm": 1.0971544981002808, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 459221187.0, "reward": 0.5546875, "reward_std": 0.10298692435026169, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1091, "tools/generated_tokens": 3889.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1163.07421875, "completions/mean_terminated_length": 1100.129638671875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.15654999669641256, "epoch": 0.18608217777493768, "frac_reward_zero_std": 0.6875, "grad_norm": 0.1864183396100998, "learning_rate": 1e-06, "loss": -0.0116, "num_tokens": 459586118.0, "reward": 0.54296875, "reward_std": 0.09617365896701813, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1092, "tools/generated_tokens": 3467.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1094.85546875, "completions/mean_terminated_length": 1064.10888671875, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.164327809587121, "epoch": 0.18625258269964, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2829054594039917, "learning_rate": 1e-06, "loss": 0.0439, "num_tokens": 459929153.0, "reward": 0.51171875, "reward_std": 0.25507354736328125, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1093, "tools/generated_tokens": 3006.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1150.984375, "completions/mean_terminated_length": 1008.923095703125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.16690885368734598, "epoch": 0.18642298762434234, "frac_reward_zero_std": 0.375, "grad_norm": 0.36384427547454834, "learning_rate": 1e-06, "loss": -0.0063, "num_tokens": 460302109.0, "reward": 0.48046875, "reward_std": 0.2525572180747986, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1094, "tools/generated_tokens": 4166.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1130.484375, "completions/mean_terminated_length": 1097.0526123046875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.15639521647244692, "epoch": 0.18659339254904467, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3211700916290283, "learning_rate": 1e-06, "loss": 0.0339, "num_tokens": 460665433.0, "reward": 0.6171875, "reward_std": 0.17749404907226562, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1095, "tools/generated_tokens": 3618.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1047.24609375, "completions/mean_terminated_length": 980.5292358398438, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.17249837517738342, "epoch": 0.186763797473747, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3065526485443115, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 461012216.0, "reward": 0.6953125, "reward_std": 0.2011812925338745, "rewards/simpleverify_reward/mean": 0.6953125, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 1096, "tools/generated_tokens": 3839.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1067.84375, "completions/mean_terminated_length": 1064.0001220703125, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.14494941849261522, "epoch": 0.18693420239844932, "frac_reward_zero_std": 0.5, "grad_norm": 0.33209049701690674, "learning_rate": 1e-06, "loss": -0.027, "num_tokens": 461356096.0, "reward": 0.5703125, "reward_std": 0.17483043670654297, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1097, "tools/generated_tokens": 3155.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1017.15234375, "completions/mean_terminated_length": 988.1766967773438, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.14074323792010546, "epoch": 0.18710460732315165, "frac_reward_zero_std": 0.375, "grad_norm": 0.34090378880500793, "learning_rate": 1e-06, "loss": -0.0101, "num_tokens": 461693831.0, "reward": 0.64453125, "reward_std": 0.24579843878746033, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1098, "tools/generated_tokens": 3385.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1134.15234375, "completions/mean_terminated_length": 1077.27392578125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.17429528757929802, "epoch": 0.18727501224785395, "frac_reward_zero_std": 0.6875, "grad_norm": 0.25587770342826843, "learning_rate": 1e-06, "loss": 0.0094, "num_tokens": 462061854.0, "reward": 0.55859375, "reward_std": 0.15042340755462646, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1099, "tools/generated_tokens": 3726.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1051.3125, "completions/mean_terminated_length": 989.278076171875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.1695699216797948, "epoch": 0.18744541717255628, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3704789876937866, "learning_rate": 1e-06, "loss": 0.0147, "num_tokens": 462414078.0, "reward": 0.5625, "reward_std": 0.29066336154937744, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1100, "tools/generated_tokens": 4379.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1083.44921875, "completions/mean_terminated_length": 955.4114990234375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.14152261801064014, "epoch": 0.1876158220972586, "frac_reward_zero_std": 0.625, "grad_norm": 0.24347573518753052, "learning_rate": 1e-06, "loss": 0.0257, "num_tokens": 462765825.0, "reward": 0.515625, "reward_std": 0.12466736882925034, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1101, "tools/generated_tokens": 3883.453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 842.5, "completions/mean_terminated_length": 833.0078735351562, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.14293293794617057, "epoch": 0.18778622702196093, "frac_reward_zero_std": 0.4375, "grad_norm": 0.35492557287216187, "learning_rate": 1e-06, "loss": -0.0101, "num_tokens": 463059169.0, "reward": 0.671875, "reward_std": 0.18859325349330902, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1102, "tools/generated_tokens": 2810.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1120.07421875, "completions/mean_terminated_length": 1001.528564453125, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.16323023941367865, "epoch": 0.18795663194666326, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3886026442050934, "learning_rate": 1e-06, "loss": 0.0223, "num_tokens": 463421716.0, "reward": 0.6171875, "reward_std": 0.23250161111354828, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1103, "tools/generated_tokens": 4080.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1119.765625, "completions/mean_terminated_length": 937.5887451171875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.15175181347876787, "epoch": 0.1881270368713656, "frac_reward_zero_std": 0.5, "grad_norm": 0.33368560671806335, "learning_rate": 1e-06, "loss": 0.0258, "num_tokens": 463807912.0, "reward": 0.4609375, "reward_std": 0.21928457915782928, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1104, "tools/generated_tokens": 4591.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1072.12890625, "completions/mean_terminated_length": 1007.0708618164062, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.14882094645872712, "epoch": 0.18829744179606792, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4135502874851227, "learning_rate": 1e-06, "loss": 0.0345, "num_tokens": 464147369.0, "reward": 0.63671875, "reward_std": 0.2100876271724701, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1105, "tools/generated_tokens": 3608.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 1999.0, "completions/mean_length": 962.78515625, "completions/mean_terminated_length": 870.8178100585938, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.1644550822675228, "epoch": 0.18846784672077024, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28991663455963135, "learning_rate": 1e-06, "loss": 0.0111, "num_tokens": 464471618.0, "reward": 0.53515625, "reward_std": 0.23593443632125854, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1106, "tools/generated_tokens": 4074.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1230.84765625, "completions/mean_terminated_length": 1122.3760986328125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.16749285906553268, "epoch": 0.18863825164547254, "frac_reward_zero_std": 0.5, "grad_norm": 0.32232341170310974, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 464857675.0, "reward": 0.359375, "reward_std": 0.2098345011472702, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1107, "tools/generated_tokens": 4278.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1023.3828125, "completions/mean_terminated_length": 990.3306274414062, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.14631283655762672, "epoch": 0.18880865657017487, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2990967333316803, "learning_rate": 1e-06, "loss": 0.0618, "num_tokens": 465205661.0, "reward": 0.53125, "reward_std": 0.17978152632713318, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1108, "tools/generated_tokens": 3783.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1133.21875, "completions/mean_terminated_length": 1029.80859375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.15191856818273664, "epoch": 0.1889790614948772, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3324718475341797, "learning_rate": 1e-06, "loss": 0.0155, "num_tokens": 465575589.0, "reward": 0.55859375, "reward_std": 0.20685213804244995, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1109, "tools/generated_tokens": 4149.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1051.40625, "completions/mean_terminated_length": 914.0977783203125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.13378596305847168, "epoch": 0.18914946641957953, "frac_reward_zero_std": 0.4375, "grad_norm": 0.33211657404899597, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 465928717.0, "reward": 0.65625, "reward_std": 0.23875632882118225, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1110, "tools/generated_tokens": 4547.40625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.70703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1096.390625, "completions/mean_terminated_length": 1006.9231567382812, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.1714788069948554, "epoch": 0.18931987134428185, "frac_reward_zero_std": 0.4375, "grad_norm": 0.30110862851142883, "learning_rate": 1e-06, "loss": 0.0352, "num_tokens": 466289553.0, "reward": 0.3359375, "reward_std": 0.22208085656166077, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 1111, "tools/generated_tokens": 3744.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1215.921875, "completions/mean_terminated_length": 966.73095703125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.1617887932807207, "epoch": 0.18949027626898418, "frac_reward_zero_std": 0.25, "grad_norm": 0.36063939332962036, "learning_rate": 1e-06, "loss": 0.0857, "num_tokens": 466683805.0, "reward": 0.59375, "reward_std": 0.2621425986289978, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1112, "tools/generated_tokens": 4855.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.77734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 908.546875, "completions/mean_terminated_length": 822.3698120117188, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.16964140813797712, "epoch": 0.1896606811936865, "frac_reward_zero_std": 0.5, "grad_norm": 0.38318413496017456, "learning_rate": 1e-06, "loss": 0.0552, "num_tokens": 467002521.0, "reward": 0.609375, "reward_std": 0.17914599180221558, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1113, "tools/generated_tokens": 3980.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1131.78515625, "completions/mean_terminated_length": 1120.9210205078125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.12496736040338874, "epoch": 0.1898310861183888, "frac_reward_zero_std": 0.5625, "grad_norm": 0.23067758977413177, "learning_rate": 1e-06, "loss": -0.0087, "num_tokens": 467347218.0, "reward": 0.75, "reward_std": 0.16218584775924683, "rewards/simpleverify_reward/mean": 0.75, "rewards/simpleverify_reward/std": 0.4338609278202057, "step": 1114, "tools/generated_tokens": 2619.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1023.875, "completions/mean_terminated_length": 927.5897827148438, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.1455818135291338, "epoch": 0.19000149104309114, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3226756155490875, "learning_rate": 1e-06, "loss": 0.0178, "num_tokens": 467682082.0, "reward": 0.58203125, "reward_std": 0.14029237627983093, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1115, "tools/generated_tokens": 3447.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1078.40234375, "completions/mean_terminated_length": 978.09912109375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.16072431951761246, "epoch": 0.19017189596779346, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4209185540676117, "learning_rate": 1e-06, "loss": 0.059, "num_tokens": 468037945.0, "reward": 0.61328125, "reward_std": 0.23308546841144562, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1116, "tools/generated_tokens": 4294.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1116.54296875, "completions/mean_terminated_length": 1050.2886962890625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.13489188207313418, "epoch": 0.1903423008924958, "frac_reward_zero_std": 0.625, "grad_norm": 0.2721402049064636, "learning_rate": 1e-06, "loss": -0.0267, "num_tokens": 468397828.0, "reward": 0.54296875, "reward_std": 0.17861157655715942, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1117, "tools/generated_tokens": 3956.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1014.88671875, "completions/mean_terminated_length": 964.0778198242188, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.13005745643749833, "epoch": 0.19051270581719812, "frac_reward_zero_std": 0.6875, "grad_norm": 0.23778405785560608, "learning_rate": 1e-06, "loss": 0.0029, "num_tokens": 468721863.0, "reward": 0.69140625, "reward_std": 0.1347825974225998, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1118, "tools/generated_tokens": 2966.89453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1993.0, "completions/mean_length": 1002.80859375, "completions/mean_terminated_length": 884.6608276367188, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.15664511267095804, "epoch": 0.19068311074190045, "frac_reward_zero_std": 0.75, "grad_norm": 0.18358772993087769, "learning_rate": 1e-06, "loss": 0.036, "num_tokens": 469055030.0, "reward": 0.421875, "reward_std": 0.08054865896701813, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1119, "tools/generated_tokens": 4042.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1008.22265625, "completions/mean_terminated_length": 970.3360595703125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.14242110447958112, "epoch": 0.19085351566660277, "frac_reward_zero_std": 0.5, "grad_norm": 0.3748657703399658, "learning_rate": 1e-06, "loss": 0.0042, "num_tokens": 469383999.0, "reward": 0.48046875, "reward_std": 0.20576362311840057, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1120, "tools/generated_tokens": 3336.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1142.22265625, "completions/mean_terminated_length": 1081.8375244140625, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.1505898702889681, "epoch": 0.1910239205913051, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3244427740573883, "learning_rate": 1e-06, "loss": 0.0299, "num_tokens": 469738808.0, "reward": 0.47265625, "reward_std": 0.19563539326190948, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1121, "tools/generated_tokens": 3854.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1157.21875, "completions/mean_terminated_length": 1097.8333740234375, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.15771446283906698, "epoch": 0.1911943255160074, "frac_reward_zero_std": 0.25, "grad_norm": 0.3382774293422699, "learning_rate": 1e-06, "loss": 0.0164, "num_tokens": 470124448.0, "reward": 0.578125, "reward_std": 0.2958957552909851, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1122, "tools/generated_tokens": 4013.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1144.12109375, "completions/mean_terminated_length": 1063.348876953125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.15331325493752956, "epoch": 0.19136473044070973, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3629094660282135, "learning_rate": 1e-06, "loss": -0.0224, "num_tokens": 470492127.0, "reward": 0.45703125, "reward_std": 0.1624389886856079, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1123, "tools/generated_tokens": 4008.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 998.38671875, "completions/mean_terminated_length": 968.8794555664062, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.14547260385006666, "epoch": 0.19153513536541206, "frac_reward_zero_std": 0.25, "grad_norm": 0.4082314372062683, "learning_rate": 1e-06, "loss": -0.0302, "num_tokens": 470822834.0, "reward": 0.609375, "reward_std": 0.3150842785835266, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1124, "tools/generated_tokens": 3526.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 989.5625, "completions/mean_terminated_length": 959.8071899414062, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.1313863224349916, "epoch": 0.19170554029011438, "frac_reward_zero_std": 0.625, "grad_norm": 0.24333229660987854, "learning_rate": 1e-06, "loss": -0.0085, "num_tokens": 471143954.0, "reward": 0.63671875, "reward_std": 0.12082062661647797, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1125, "tools/generated_tokens": 3253.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1138.1875, "completions/mean_terminated_length": 1012.8355712890625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.16582213435322046, "epoch": 0.1918759452148167, "frac_reward_zero_std": 0.625, "grad_norm": 0.2747613787651062, "learning_rate": 1e-06, "loss": -0.0021, "num_tokens": 471520610.0, "reward": 0.48046875, "reward_std": 0.15778234601020813, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1126, "tools/generated_tokens": 4706.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1066.1796875, "completions/mean_terminated_length": 945.6052856445312, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.14380132034420967, "epoch": 0.19204635013951904, "frac_reward_zero_std": 0.25, "grad_norm": 0.30910858511924744, "learning_rate": 1e-06, "loss": -0.0323, "num_tokens": 471870704.0, "reward": 0.60546875, "reward_std": 0.2425214648246765, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1127, "tools/generated_tokens": 4314.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1013.1015625, "completions/mean_terminated_length": 979.7177124023438, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.1407800940796733, "epoch": 0.19221675506422137, "frac_reward_zero_std": 0.5, "grad_norm": 0.3097638785839081, "learning_rate": 1e-06, "loss": -0.0214, "num_tokens": 472198282.0, "reward": 0.46484375, "reward_std": 0.1871562898159027, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1128, "tools/generated_tokens": 3245.1015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.08984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1143.0, "completions/mean_terminated_length": 1031.859619140625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.17472115065902472, "epoch": 0.19238715998892367, "frac_reward_zero_std": 0.4375, "grad_norm": 0.313220351934433, "learning_rate": 1e-06, "loss": -0.0286, "num_tokens": 472568938.0, "reward": 0.484375, "reward_std": 0.23369672894477844, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1129, "tools/generated_tokens": 4478.99609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 1999.0, "completions/mean_length": 1009.24609375, "completions/mean_terminated_length": 958.1597900390625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.15969266649335623, "epoch": 0.192557564913626, "frac_reward_zero_std": 0.5, "grad_norm": 0.3700685501098633, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 472906649.0, "reward": 0.5234375, "reward_std": 0.19760413467884064, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1130, "tools/generated_tokens": 3769.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1132.09765625, "completions/mean_terminated_length": 1045.9871826171875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.15304421912878752, "epoch": 0.19272796983832832, "frac_reward_zero_std": 0.375, "grad_norm": 0.3774626553058624, "learning_rate": 1e-06, "loss": 0.0169, "num_tokens": 473272594.0, "reward": 0.5234375, "reward_std": 0.18683473765850067, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1131, "tools/generated_tokens": 4124.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1135.65234375, "completions/mean_terminated_length": 1066.6513671875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.17038383055478334, "epoch": 0.19289837476303065, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4009467661380768, "learning_rate": 1e-06, "loss": 0.0195, "num_tokens": 473646841.0, "reward": 0.5546875, "reward_std": 0.26649209856987, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1132, "tools/generated_tokens": 4527.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1109.43359375, "completions/mean_terminated_length": 1042.673583984375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.145938606467098, "epoch": 0.19306877968773298, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2797616124153137, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 474009624.0, "reward": 0.5546875, "reward_std": 0.2170758694410324, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1133, "tools/generated_tokens": 3629.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1005.98046875, "completions/mean_terminated_length": 980.9720458984375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.15513051208108664, "epoch": 0.1932391846124353, "frac_reward_zero_std": 0.625, "grad_norm": 0.3377520740032196, "learning_rate": 1e-06, "loss": 0.011, "num_tokens": 474344691.0, "reward": 0.6171875, "reward_std": 0.1666867583990097, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1134, "tools/generated_tokens": 3525.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1051.00390625, "completions/mean_terminated_length": 1014.6761474609375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.16334208194166422, "epoch": 0.19340958953713763, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4066193997859955, "learning_rate": 1e-06, "loss": -0.0255, "num_tokens": 474697380.0, "reward": 0.51171875, "reward_std": 0.2889442443847656, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1135, "tools/generated_tokens": 4467.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2006.0, "completions/mean_length": 1032.3359375, "completions/mean_terminated_length": 1003.7830810546875, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.14251257292926311, "epoch": 0.19357999446183996, "frac_reward_zero_std": 0.5, "grad_norm": 0.6139031648635864, "learning_rate": 1e-06, "loss": -0.0289, "num_tokens": 475034474.0, "reward": 0.5859375, "reward_std": 0.18739622831344604, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1136, "tools/generated_tokens": 3296.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1056.75, "completions/mean_terminated_length": 999.4049072265625, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.1657526195049286, "epoch": 0.19375039938654226, "frac_reward_zero_std": 0.5, "grad_norm": 0.45410311222076416, "learning_rate": 1e-06, "loss": -0.0214, "num_tokens": 475389002.0, "reward": 0.40625, "reward_std": 0.2129075825214386, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1137, "tools/generated_tokens": 3824.74609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1067.21875, "completions/mean_terminated_length": 1063.37255859375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.12899543344974518, "epoch": 0.1939208043112446, "frac_reward_zero_std": 0.5, "grad_norm": 0.37450122833251953, "learning_rate": 1e-06, "loss": -0.0323, "num_tokens": 475733810.0, "reward": 0.58203125, "reward_std": 0.21818022429943085, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1138, "tools/generated_tokens": 3043.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1012.89453125, "completions/mean_terminated_length": 966.4203491210938, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.18418116681277752, "epoch": 0.19409120923594692, "frac_reward_zero_std": 0.4375, "grad_norm": 0.48091068863868713, "learning_rate": 1e-06, "loss": -0.0084, "num_tokens": 476091191.0, "reward": 0.3515625, "reward_std": 0.24149677157402039, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1139, "tools/generated_tokens": 4564.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1033.4765625, "completions/mean_terminated_length": 996.5101318359375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.14348224271088839, "epoch": 0.19426161416064924, "frac_reward_zero_std": 0.625, "grad_norm": 0.4966225326061249, "learning_rate": 1e-06, "loss": -0.005, "num_tokens": 476429169.0, "reward": 0.50390625, "reward_std": 0.13699321448802948, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1140, "tools/generated_tokens": 3961.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 956.87109375, "completions/mean_terminated_length": 859.3659057617188, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.1712651178240776, "epoch": 0.19443201908535157, "frac_reward_zero_std": 0.4375, "grad_norm": 0.32756173610687256, "learning_rate": 1e-06, "loss": -0.0155, "num_tokens": 476760000.0, "reward": 0.484375, "reward_std": 0.1643964648246765, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1141, "tools/generated_tokens": 4404.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 947.79296875, "completions/mean_terminated_length": 930.3294067382812, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.14397554891183972, "epoch": 0.1946024240100539, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3297126293182373, "learning_rate": 1e-06, "loss": -0.0212, "num_tokens": 477081579.0, "reward": 0.65234375, "reward_std": 0.23007602989673615, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1142, "tools/generated_tokens": 3531.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 974.359375, "completions/mean_terminated_length": 948.592041015625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.15604246128350496, "epoch": 0.19477282893475623, "frac_reward_zero_std": 0.375, "grad_norm": 0.2804981470108032, "learning_rate": 1e-06, "loss": -0.0163, "num_tokens": 477403863.0, "reward": 0.53125, "reward_std": 0.20630648732185364, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1143, "tools/generated_tokens": 3614.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2045.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 815.65625, "completions/mean_terminated_length": 815.65625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.15150599228218198, "epoch": 0.19494323385945853, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3885502517223358, "learning_rate": 1e-06, "loss": -0.041, "num_tokens": 477680367.0, "reward": 0.41015625, "reward_std": 0.170307457447052, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1144, "tools/generated_tokens": 3167.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 949.12890625, "completions/mean_terminated_length": 880.7344970703125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.12261653458699584, "epoch": 0.19511363878416085, "frac_reward_zero_std": 0.625, "grad_norm": 0.3100487291812897, "learning_rate": 1e-06, "loss": 0.0077, "num_tokens": 477992816.0, "reward": 0.6875, "reward_std": 0.13149453699588776, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1145, "tools/generated_tokens": 3189.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 987.82421875, "completions/mean_terminated_length": 970.99609375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.15273478347808123, "epoch": 0.19528404370886318, "frac_reward_zero_std": 0.375, "grad_norm": 0.41958087682724, "learning_rate": 1e-06, "loss": -0.025, "num_tokens": 478324947.0, "reward": 0.37109375, "reward_std": 0.19641819596290588, "rewards/simpleverify_reward/mean": 0.37109375, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1146, "tools/generated_tokens": 4099.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1001.41796875, "completions/mean_terminated_length": 989.0079345703125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.14745144452899694, "epoch": 0.1954544486335655, "frac_reward_zero_std": 0.375, "grad_norm": 0.40877947211265564, "learning_rate": 1e-06, "loss": 0.0419, "num_tokens": 478644238.0, "reward": 0.6875, "reward_std": 0.2531684637069702, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1147, "tools/generated_tokens": 3209.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1054.35546875, "completions/mean_terminated_length": 1022.3023681640625, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.16118179354816675, "epoch": 0.19562485355826784, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3094380497932434, "learning_rate": 1e-06, "loss": -0.0187, "num_tokens": 478990249.0, "reward": 0.42578125, "reward_std": 0.18880629539489746, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1148, "tools/generated_tokens": 4062.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 943.08203125, "completions/mean_terminated_length": 929.9802856445312, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.157478095497936, "epoch": 0.19579525848297016, "frac_reward_zero_std": 0.5, "grad_norm": 0.3753894865512848, "learning_rate": 1e-06, "loss": -0.0195, "num_tokens": 479315006.0, "reward": 0.46484375, "reward_std": 0.16926807165145874, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1149, "tools/generated_tokens": 3527.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 990.08984375, "completions/mean_terminated_length": 985.9412231445312, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.17706937342882156, "epoch": 0.1959656634076725, "frac_reward_zero_std": 0.375, "grad_norm": 0.44601136445999146, "learning_rate": 1e-06, "loss": -0.0373, "num_tokens": 479649205.0, "reward": 0.55078125, "reward_std": 0.24111157655715942, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1150, "tools/generated_tokens": 3822.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1118.66796875, "completions/mean_terminated_length": 1084.8056640625, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.13297047093510628, "epoch": 0.19613606833237482, "frac_reward_zero_std": 0.5625, "grad_norm": 0.26272010803222656, "learning_rate": 1e-06, "loss": -0.0062, "num_tokens": 479997616.0, "reward": 0.703125, "reward_std": 0.19116157293319702, "rewards/simpleverify_reward/mean": 0.703125, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 1151, "tools/generated_tokens": 3206.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1033.2734375, "completions/mean_terminated_length": 978.9876098632812, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.15683725383132696, "epoch": 0.19630647325707712, "frac_reward_zero_std": 0.4375, "grad_norm": 0.40224725008010864, "learning_rate": 1e-06, "loss": 0.0614, "num_tokens": 480340198.0, "reward": 0.578125, "reward_std": 0.22060389816761017, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1152, "tools/generated_tokens": 3753.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 952.9765625, "completions/mean_terminated_length": 935.5952758789062, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.1648413985967636, "epoch": 0.19647687818177945, "frac_reward_zero_std": 0.4375, "grad_norm": 0.41126322746276855, "learning_rate": 1e-06, "loss": 0.0031, "num_tokens": 480653696.0, "reward": 0.5859375, "reward_std": 0.18188363313674927, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1153, "tools/generated_tokens": 3232.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1004.56640625, "completions/mean_terminated_length": 975.23291015625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.18359979800879955, "epoch": 0.19664728310648177, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3361515402793884, "learning_rate": 1e-06, "loss": -0.0338, "num_tokens": 480979697.0, "reward": 0.55078125, "reward_std": 0.1446847915649414, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1154, "tools/generated_tokens": 3852.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 942.46875, "completions/mean_terminated_length": 938.1333618164062, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.16289528366178274, "epoch": 0.1968176880311841, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5754010081291199, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 481305161.0, "reward": 0.5859375, "reward_std": 0.26920706033706665, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1155, "tools/generated_tokens": 3854.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 941.8203125, "completions/mean_terminated_length": 937.482421875, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.14824689086526632, "epoch": 0.19698809295588643, "frac_reward_zero_std": 0.5625, "grad_norm": 0.402113139629364, "learning_rate": 1e-06, "loss": -0.0257, "num_tokens": 481621883.0, "reward": 0.6015625, "reward_std": 0.174540713429451, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1156, "tools/generated_tokens": 2957.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1008.765625, "completions/mean_terminated_length": 988.0637817382812, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.1683209352195263, "epoch": 0.19715849788058876, "frac_reward_zero_std": 0.5625, "grad_norm": 0.36307811737060547, "learning_rate": 1e-06, "loss": -0.0292, "num_tokens": 481954447.0, "reward": 0.4375, "reward_std": 0.15261822938919067, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1157, "tools/generated_tokens": 3632.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 805.07421875, "completions/mean_terminated_length": 800.2000732421875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.18004435300827026, "epoch": 0.19732890280529108, "frac_reward_zero_std": 0.5, "grad_norm": 0.36233004927635193, "learning_rate": 1e-06, "loss": 0.0086, "num_tokens": 482244914.0, "reward": 0.44921875, "reward_std": 0.2015429437160492, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1158, "tools/generated_tokens": 3421.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1039.84375, "completions/mean_terminated_length": 1019.760986328125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.16570425685495138, "epoch": 0.19749930772999338, "frac_reward_zero_std": 0.6875, "grad_norm": 0.24720041453838348, "learning_rate": 1e-06, "loss": 0.0089, "num_tokens": 482574154.0, "reward": 0.6328125, "reward_std": 0.09814241528511047, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1159, "tools/generated_tokens": 3119.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 914.9921875, "completions/mean_terminated_length": 914.9921875, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1529004625044763, "epoch": 0.1976697126546957, "frac_reward_zero_std": 0.4375, "grad_norm": 0.47393178939819336, "learning_rate": 1e-06, "loss": -0.0238, "num_tokens": 482886552.0, "reward": 0.60546875, "reward_std": 0.19893452525138855, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1160, "tools/generated_tokens": 3362.99609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 941.3203125, "completions/mean_terminated_length": 941.3203125, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.15938566625118256, "epoch": 0.19784011757939804, "frac_reward_zero_std": 0.5, "grad_norm": 0.8478848338127136, "learning_rate": 1e-06, "loss": 0.0363, "num_tokens": 483196970.0, "reward": 0.734375, "reward_std": 0.19554270803928375, "rewards/simpleverify_reward/mean": 0.734375, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 1161, "tools/generated_tokens": 2749.31640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1009.45703125, "completions/mean_terminated_length": 992.9722900390625, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.14677808713167906, "epoch": 0.19801052250410037, "frac_reward_zero_std": 0.6875, "grad_norm": 0.21593178808689117, "learning_rate": 1e-06, "loss": -0.0175, "num_tokens": 483523247.0, "reward": 0.58203125, "reward_std": 0.13643454015254974, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1162, "tools/generated_tokens": 3289.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 900.2109375, "completions/mean_terminated_length": 895.7098388671875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.16765887010842562, "epoch": 0.1981809274288027, "frac_reward_zero_std": 0.5, "grad_norm": 0.4278518259525299, "learning_rate": 1e-06, "loss": -0.0199, "num_tokens": 483833333.0, "reward": 0.61328125, "reward_std": 0.20576170086860657, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1163, "tools/generated_tokens": 3204.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 913.171875, "completions/mean_terminated_length": 908.7216186523438, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.1799656106159091, "epoch": 0.19835133235350502, "frac_reward_zero_std": 0.6875, "grad_norm": 0.25058895349502563, "learning_rate": 1e-06, "loss": 0.0109, "num_tokens": 484144561.0, "reward": 0.515625, "reward_std": 0.13149453699588776, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1164, "tools/generated_tokens": 3577.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2044.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 960.3828125, "completions/mean_terminated_length": 960.3828125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.1431764136068523, "epoch": 0.19852173727820735, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3615739643573761, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 484469059.0, "reward": 0.7421875, "reward_std": 0.16878889501094818, "rewards/simpleverify_reward/mean": 0.7421875, "rewards/simpleverify_reward/std": 0.4382871091365814, "step": 1165, "tools/generated_tokens": 3064.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.02734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1000.88671875, "completions/mean_terminated_length": 996.7804565429688, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1888198684900999, "epoch": 0.19869214220290968, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3677363991737366, "learning_rate": 1e-06, "loss": -0.0282, "num_tokens": 484793654.0, "reward": 0.53125, "reward_std": 0.17231407761573792, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1166, "tools/generated_tokens": 3320.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2036.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1002.390625, "completions/mean_terminated_length": 1002.390625, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.15965775586664677, "epoch": 0.19886254712761198, "frac_reward_zero_std": 0.625, "grad_norm": 0.2893698513507843, "learning_rate": 1e-06, "loss": -0.0084, "num_tokens": 485130458.0, "reward": 0.546875, "reward_std": 0.13912242650985718, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1167, "tools/generated_tokens": 3266.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1011.9140625, "completions/mean_terminated_length": 999.6284790039062, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.17127364501357079, "epoch": 0.1990329520523143, "frac_reward_zero_std": 0.3125, "grad_norm": 0.37185758352279663, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 485472196.0, "reward": 0.65234375, "reward_std": 0.2819339632987976, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1168, "tools/generated_tokens": 3611.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 921.85546875, "completions/mean_terminated_length": 908.5020141601562, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.18364347517490387, "epoch": 0.19920335697701663, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4362296760082245, "learning_rate": 1e-06, "loss": -0.0426, "num_tokens": 485786735.0, "reward": 0.640625, "reward_std": 0.2404082715511322, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1169, "tools/generated_tokens": 3881.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2036.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1037.5859375, "completions/mean_terminated_length": 1037.5859375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.149854289367795, "epoch": 0.19937376190171896, "frac_reward_zero_std": 0.5, "grad_norm": 0.3755163848400116, "learning_rate": 1e-06, "loss": 0.0014, "num_tokens": 486115429.0, "reward": 0.58984375, "reward_std": 0.19135859608650208, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1170, "tools/generated_tokens": 2773.5859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.84765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 950.98828125, "completions/mean_terminated_length": 942.3504028320312, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.17896776646375656, "epoch": 0.1995441668264213, "frac_reward_zero_std": 0.625, "grad_norm": 0.2880171835422516, "learning_rate": 1e-06, "loss": -0.0298, "num_tokens": 486437842.0, "reward": 0.375, "reward_std": 0.11949022114276886, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1171, "tools/generated_tokens": 3567.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1044.0078125, "completions/mean_terminated_length": 1024.008056640625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.13886642642319202, "epoch": 0.19971457175112362, "frac_reward_zero_std": 0.75, "grad_norm": 0.20382791757583618, "learning_rate": 1e-06, "loss": 0.0212, "num_tokens": 486769332.0, "reward": 0.6640625, "reward_std": 0.10906945914030075, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 1172, "tools/generated_tokens": 3124.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2028.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 911.3984375, "completions/mean_terminated_length": 911.3984375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.1530983718112111, "epoch": 0.19988497667582594, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5948839783668518, "learning_rate": 1e-06, "loss": -0.0311, "num_tokens": 487068250.0, "reward": 0.59375, "reward_std": 0.25263670086860657, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1173, "tools/generated_tokens": 2655.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 999.75, "completions/mean_terminated_length": 991.4960327148438, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.17054700199514627, "epoch": 0.20005538160052824, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2992614209651947, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 487395322.0, "reward": 0.52734375, "reward_std": 0.161190003156662, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1174, "tools/generated_tokens": 3519.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1029.46875, "completions/mean_terminated_length": 1005.0240478515625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.16317198891192675, "epoch": 0.20022578652523057, "frac_reward_zero_std": 0.375, "grad_norm": 0.3809545636177063, "learning_rate": 1e-06, "loss": -0.0378, "num_tokens": 487734770.0, "reward": 0.54296875, "reward_std": 0.23502713441848755, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1175, "tools/generated_tokens": 3709.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1075.1328125, "completions/mean_terminated_length": 1067.472412109375, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.1667528934776783, "epoch": 0.2003961914499329, "frac_reward_zero_std": 0.625, "grad_norm": 0.3168644309043884, "learning_rate": 1e-06, "loss": 0.0259, "num_tokens": 488073972.0, "reward": 0.66015625, "reward_std": 0.1556118279695511, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1176, "tools/generated_tokens": 2763.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 912.30859375, "completions/mean_terminated_length": 907.85498046875, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.16962833981961012, "epoch": 0.20056659637463523, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3182103633880615, "learning_rate": 1e-06, "loss": 0.0165, "num_tokens": 488385859.0, "reward": 0.64453125, "reward_std": 0.18486879765987396, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1177, "tools/generated_tokens": 3456.3046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2012.0, "completions/mean_length": 1005.98046875, "completions/mean_terminated_length": 985.22314453125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.1710472172126174, "epoch": 0.20073700129933755, "frac_reward_zero_std": 0.5, "grad_norm": 0.41216591000556946, "learning_rate": 1e-06, "loss": -0.0142, "num_tokens": 488714158.0, "reward": 0.55859375, "reward_std": 0.17154711484909058, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1178, "tools/generated_tokens": 3333.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 930.3984375, "completions/mean_terminated_length": 926.0157470703125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.17989927250891924, "epoch": 0.20090740622403988, "frac_reward_zero_std": 0.375, "grad_norm": 0.4357042908668518, "learning_rate": 1e-06, "loss": 0.0883, "num_tokens": 489028900.0, "reward": 0.5859375, "reward_std": 0.262538880109787, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1179, "tools/generated_tokens": 3594.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1052.34375, "completions/mean_terminated_length": 1044.50390625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.15615292824804783, "epoch": 0.2010778111487422, "frac_reward_zero_std": 0.625, "grad_norm": 0.2749071419239044, "learning_rate": 1e-06, "loss": -0.0285, "num_tokens": 489361788.0, "reward": 0.61328125, "reward_std": 0.17114414274692535, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1180, "tools/generated_tokens": 3356.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1181.15234375, "completions/mean_terminated_length": 1181.15234375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.15004483610391617, "epoch": 0.20124821607344454, "frac_reward_zero_std": 0.375, "grad_norm": 0.2822796702384949, "learning_rate": 1e-06, "loss": -0.0065, "num_tokens": 489719363.0, "reward": 0.4921875, "reward_std": 0.20960843563079834, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1181, "tools/generated_tokens": 2565.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.67578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 963.7109375, "completions/mean_terminated_length": 959.4588623046875, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.17817926779389381, "epoch": 0.20141862099814684, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4142369031906128, "learning_rate": 1e-06, "loss": -0.0138, "num_tokens": 490046329.0, "reward": 0.57421875, "reward_std": 0.1975114494562149, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1182, "tools/generated_tokens": 3443.703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2027.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1031.109375, "completions/mean_terminated_length": 1031.109375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.14635597821325064, "epoch": 0.20158902592284916, "frac_reward_zero_std": 0.75, "grad_norm": 0.21157822012901306, "learning_rate": 1e-06, "loss": 0.005, "num_tokens": 490376005.0, "reward": 0.47265625, "reward_std": 0.08912044763565063, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1183, "tools/generated_tokens": 2991.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1028.19921875, "completions/mean_terminated_length": 973.6419677734375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.17967787384986877, "epoch": 0.2017594308475515, "frac_reward_zero_std": 0.5, "grad_norm": 0.2967304587364197, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 490712328.0, "reward": 0.59765625, "reward_std": 0.20883671939373016, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1184, "tools/generated_tokens": 3780.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1108.38671875, "completions/mean_terminated_length": 1089.67333984375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1862321263179183, "epoch": 0.20192983577225382, "frac_reward_zero_std": 0.5, "grad_norm": 0.2902344763278961, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 491072939.0, "reward": 0.46484375, "reward_std": 0.19188132882118225, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1185, "tools/generated_tokens": 3948.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1021.671875, "completions/mean_terminated_length": 1021.671875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.1579496068879962, "epoch": 0.20210024069695615, "frac_reward_zero_std": 0.25, "grad_norm": 0.46387264132499695, "learning_rate": 1e-06, "loss": 0.0264, "num_tokens": 491412151.0, "reward": 0.5078125, "reward_std": 0.2837800979614258, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1186, "tools/generated_tokens": 3229.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 946.0703125, "completions/mean_terminated_length": 937.3936767578125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.19036847539246082, "epoch": 0.20227064562165847, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4464570879936218, "learning_rate": 1e-06, "loss": 0.003, "num_tokens": 491723209.0, "reward": 0.46484375, "reward_std": 0.18157809972763062, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1187, "tools/generated_tokens": 3442.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1052.1875, "completions/mean_terminated_length": 1044.346435546875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.16243121400475502, "epoch": 0.2024410505463608, "frac_reward_zero_std": 0.5, "grad_norm": 0.36912113428115845, "learning_rate": 1e-06, "loss": 0.0221, "num_tokens": 492060777.0, "reward": 0.7265625, "reward_std": 0.19321171939373016, "rewards/simpleverify_reward/mean": 0.7265625, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 1188, "tools/generated_tokens": 3164.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.03125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1088.89453125, "completions/mean_terminated_length": 1073.670654296875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.1862728465348482, "epoch": 0.2026114554710631, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3206484317779541, "learning_rate": 1e-06, "loss": -0.0287, "num_tokens": 492427438.0, "reward": 0.625, "reward_std": 0.18297690153121948, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1189, "tools/generated_tokens": 3984.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1064.41015625, "completions/mean_terminated_length": 1056.6654052734375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.170456369407475, "epoch": 0.20278186039576543, "frac_reward_zero_std": 0.4375, "grad_norm": 0.34780028462409973, "learning_rate": 1e-06, "loss": 0.0043, "num_tokens": 492773319.0, "reward": 0.57421875, "reward_std": 0.21060428023338318, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1190, "tools/generated_tokens": 3424.4140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1040.44140625, "completions/mean_terminated_length": 1036.490234375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.17261014878749847, "epoch": 0.20295226532046776, "frac_reward_zero_std": 0.3125, "grad_norm": 0.32785069942474365, "learning_rate": 1e-06, "loss": 0.0125, "num_tokens": 493118024.0, "reward": 0.58203125, "reward_std": 0.23315106332302094, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1191, "tools/generated_tokens": 3384.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1088.17578125, "completions/mean_terminated_length": 1084.411865234375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.16779708955436945, "epoch": 0.20312267024517008, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2641289532184601, "learning_rate": 1e-06, "loss": -0.0335, "num_tokens": 493465029.0, "reward": 0.69140625, "reward_std": 0.10881631821393967, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1192, "tools/generated_tokens": 3080.171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 957.4921875, "completions/mean_terminated_length": 957.4921875, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.16718083806335926, "epoch": 0.2032930751698724, "frac_reward_zero_std": 0.5, "grad_norm": 0.29929691553115845, "learning_rate": 1e-06, "loss": 0.001, "num_tokens": 493789411.0, "reward": 0.59375, "reward_std": 0.2112603783607483, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1193, "tools/generated_tokens": 3109.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1037.359375, "completions/mean_terminated_length": 1033.3961181640625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.15680260118097067, "epoch": 0.20346348009457474, "frac_reward_zero_std": 0.625, "grad_norm": 0.21234369277954102, "learning_rate": 1e-06, "loss": -0.0158, "num_tokens": 494120655.0, "reward": 0.578125, "reward_std": 0.14578913152217865, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1194, "tools/generated_tokens": 2637.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2007.0, "completions/mean_length": 909.65625, "completions/mean_terminated_length": 886.9840698242188, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.16557593271136284, "epoch": 0.20363388501927707, "frac_reward_zero_std": 0.25, "grad_norm": 0.4424571394920349, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 494419719.0, "reward": 0.65234375, "reward_std": 0.271121621131897, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1195, "tools/generated_tokens": 3053.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1086.01171875, "completions/mean_terminated_length": 1021.8792114257812, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.21148150693625212, "epoch": 0.2038042899439794, "frac_reward_zero_std": 0.5, "grad_norm": 0.2902919054031372, "learning_rate": 1e-06, "loss": 0.0609, "num_tokens": 494784282.0, "reward": 0.2265625, "reward_std": 0.18541166186332703, "rewards/simpleverify_reward/mean": 0.2265625, "rewards/simpleverify_reward/std": 0.41942715644836426, "step": 1196, "tools/generated_tokens": 4686.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1082.18359375, "completions/mean_terminated_length": 1042.9227294921875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.1748366253450513, "epoch": 0.2039746948686817, "frac_reward_zero_std": 0.625, "grad_norm": 0.2821313440799713, "learning_rate": 1e-06, "loss": -0.0128, "num_tokens": 495126473.0, "reward": 0.6015625, "reward_std": 0.1468139886856079, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1197, "tools/generated_tokens": 3370.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 993.1015625, "completions/mean_terminated_length": 967.7840576171875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.17694676481187344, "epoch": 0.20414509979338402, "frac_reward_zero_std": 0.5625, "grad_norm": 0.27799010276794434, "learning_rate": 1e-06, "loss": 0.006, "num_tokens": 495457555.0, "reward": 0.5546875, "reward_std": 0.18211251497268677, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1198, "tools/generated_tokens": 3713.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1088.40234375, "completions/mean_terminated_length": 1069.286865234375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.18707407638430595, "epoch": 0.20431550471808635, "frac_reward_zero_std": 0.5625, "grad_norm": 0.40969178080558777, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 495813578.0, "reward": 0.6875, "reward_std": 0.1673629879951477, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1199, "tools/generated_tokens": 3024.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1077.76171875, "completions/mean_terminated_length": 1046.463623046875, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.19125044997781515, "epoch": 0.20448590964278868, "frac_reward_zero_std": 0.375, "grad_norm": 0.3800821602344513, "learning_rate": 1e-06, "loss": -0.0326, "num_tokens": 496164525.0, "reward": 0.67578125, "reward_std": 0.2255660891532898, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1200, "tools/generated_tokens": 3605.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1031.9296875, "completions/mean_terminated_length": 1019.8814697265625, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.19392258767038584, "epoch": 0.204656314567491, "frac_reward_zero_std": 0.375, "grad_norm": 0.3359970450401306, "learning_rate": 1e-06, "loss": 0.0493, "num_tokens": 496503563.0, "reward": 0.6796875, "reward_std": 0.2625119686126709, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1201, "tools/generated_tokens": 3663.9296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1157.76171875, "completions/mean_terminated_length": 1147.20556640625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.18044502288103104, "epoch": 0.20482671949219333, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3041244149208069, "learning_rate": 1e-06, "loss": 0.0358, "num_tokens": 496864942.0, "reward": 0.56640625, "reward_std": 0.24789774417877197, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1202, "tools/generated_tokens": 3493.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1121.70703125, "completions/mean_terminated_length": 1076.151611328125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.1809568451717496, "epoch": 0.20499712441689566, "frac_reward_zero_std": 0.375, "grad_norm": 0.3332161605358124, "learning_rate": 1e-06, "loss": 0.004, "num_tokens": 497223187.0, "reward": 0.62890625, "reward_std": 0.2284260094165802, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1203, "tools/generated_tokens": 3793.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1035.0546875, "completions/mean_terminated_length": 976.4545288085938, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.19257738161832094, "epoch": 0.20516752934159796, "frac_reward_zero_std": 0.5, "grad_norm": 0.3475959897041321, "learning_rate": 1e-06, "loss": -0.0023, "num_tokens": 497572257.0, "reward": 0.52734375, "reward_std": 0.19068430364131927, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1204, "tools/generated_tokens": 4227.05078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1035.6640625, "completions/mean_terminated_length": 1007.2047729492188, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.18402176909148693, "epoch": 0.2053379342663003, "frac_reward_zero_std": 0.5625, "grad_norm": 0.29811468720436096, "learning_rate": 1e-06, "loss": 0.0086, "num_tokens": 497907131.0, "reward": 0.50390625, "reward_std": 0.15931200981140137, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1205, "tools/generated_tokens": 3739.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2044.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1018.890625, "completions/mean_terminated_length": 1018.890625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.20332183223217726, "epoch": 0.20550833919100261, "frac_reward_zero_std": 0.625, "grad_norm": 0.2540242373943329, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 498236751.0, "reward": 0.53515625, "reward_std": 0.12333696335554123, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1206, "tools/generated_tokens": 3298.89453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 971.37109375, "completions/mean_terminated_length": 967.1490478515625, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.1817765338346362, "epoch": 0.20567874411570494, "frac_reward_zero_std": 0.5625, "grad_norm": 0.28091898560523987, "learning_rate": 1e-06, "loss": -0.0113, "num_tokens": 498560014.0, "reward": 0.5234375, "reward_std": 0.18001039326190948, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1207, "tools/generated_tokens": 3435.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1110.32421875, "completions/mean_terminated_length": 1080.0765380859375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.19555593840777874, "epoch": 0.20584914904040727, "frac_reward_zero_std": 0.5, "grad_norm": 0.26679936051368713, "learning_rate": 1e-06, "loss": 0.0009, "num_tokens": 498912705.0, "reward": 0.39453125, "reward_std": 0.215663880109787, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1208, "tools/generated_tokens": 3806.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1101.796875, "completions/mean_terminated_length": 1047.057861328125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.18645263556391, "epoch": 0.2060195539651096, "frac_reward_zero_std": 0.5, "grad_norm": 0.24799497425556183, "learning_rate": 1e-06, "loss": 0.0035, "num_tokens": 499261821.0, "reward": 0.4140625, "reward_std": 0.1955379694700241, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1209, "tools/generated_tokens": 3621.796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1259.1796875, "completions/mean_terminated_length": 1210.0830078125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.17355221416801214, "epoch": 0.20618995888981193, "frac_reward_zero_std": 0.5, "grad_norm": 0.33726832270622253, "learning_rate": 1e-06, "loss": 0.0025, "num_tokens": 499655579.0, "reward": 0.36328125, "reward_std": 0.21048866212368011, "rewards/simpleverify_reward/mean": 0.36328125, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1210, "tools/generated_tokens": 3579.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 985.0859375, "completions/mean_terminated_length": 880.1630859375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.22012237086892128, "epoch": 0.20636036381451425, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2119995355606079, "learning_rate": 1e-06, "loss": 0.0166, "num_tokens": 499997745.0, "reward": 0.47265625, "reward_std": 0.1438203752040863, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1211, "tools/generated_tokens": 4817.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 999.515625, "completions/mean_terminated_length": 974.35205078125, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.17588932532817125, "epoch": 0.20653076873921655, "frac_reward_zero_std": 0.5, "grad_norm": 0.2990218997001648, "learning_rate": 1e-06, "loss": 0.017, "num_tokens": 500334901.0, "reward": 0.57421875, "reward_std": 0.19496729969978333, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1212, "tools/generated_tokens": 3623.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1118.47265625, "completions/mean_terminated_length": 1052.3555908203125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.21204277407377958, "epoch": 0.20670117366391888, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4580848217010498, "learning_rate": 1e-06, "loss": -0.0001, "num_tokens": 500705934.0, "reward": 0.40625, "reward_std": 0.25372907519340515, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1213, "tools/generated_tokens": 4574.4765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1138.0625, "completions/mean_terminated_length": 1104.9068603515625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.1884017651900649, "epoch": 0.2068715785886212, "frac_reward_zero_std": 0.1875, "grad_norm": 0.5224053263664246, "learning_rate": 1e-06, "loss": -0.022, "num_tokens": 501070622.0, "reward": 0.65625, "reward_std": 0.281544029712677, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1214, "tools/generated_tokens": 3690.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1195.37890625, "completions/mean_terminated_length": 1134.732177734375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.1815424906089902, "epoch": 0.20704198351332354, "frac_reward_zero_std": 0.625, "grad_norm": 0.36153945326805115, "learning_rate": 1e-06, "loss": -0.0312, "num_tokens": 501451935.0, "reward": 0.546875, "reward_std": 0.13093778491020203, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1215, "tools/generated_tokens": 3811.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1076.5078125, "completions/mean_terminated_length": 1053.1920166015625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.17570126056671143, "epoch": 0.20721238843802586, "frac_reward_zero_std": 0.5625, "grad_norm": 0.25938552618026733, "learning_rate": 1e-06, "loss": -0.0087, "num_tokens": 501799473.0, "reward": 0.55078125, "reward_std": 0.16746041178703308, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1216, "tools/generated_tokens": 3244.5, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1022.91015625, "completions/mean_terminated_length": 994.0923461914062, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.18576069548726082, "epoch": 0.2073827933627282, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3006632626056671, "learning_rate": 1e-06, "loss": 0.05, "num_tokens": 502136458.0, "reward": 0.5078125, "reward_std": 0.22588762640953064, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1217, "tools/generated_tokens": 3830.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1054.3046875, "completions/mean_terminated_length": 1013.9105224609375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.18343947548419237, "epoch": 0.20755319828743052, "frac_reward_zero_std": 0.5, "grad_norm": 0.3299502730369568, "learning_rate": 1e-06, "loss": -0.0014, "num_tokens": 502465944.0, "reward": 0.44140625, "reward_std": 0.1701192855834961, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1218, "tools/generated_tokens": 3334.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 917.31640625, "completions/mean_terminated_length": 908.4133911132812, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.18713217414915562, "epoch": 0.20772360321213282, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2960425913333893, "learning_rate": 1e-06, "loss": -0.0222, "num_tokens": 502773641.0, "reward": 0.69140625, "reward_std": 0.16266503930091858, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1219, "tools/generated_tokens": 3181.32421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 990.2734375, "completions/mean_terminated_length": 885.8626708984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.18949238676577806, "epoch": 0.20789400813683515, "frac_reward_zero_std": 0.5, "grad_norm": 0.26516106724739075, "learning_rate": 1e-06, "loss": 0.0128, "num_tokens": 503102767.0, "reward": 0.48046875, "reward_std": 0.2176070213317871, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1220, "tools/generated_tokens": 4254.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1104.859375, "completions/mean_terminated_length": 1016.1923828125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.20044037234038115, "epoch": 0.20806441306153747, "frac_reward_zero_std": 0.25, "grad_norm": 0.37143674492836, "learning_rate": 1e-06, "loss": -0.0079, "num_tokens": 503473371.0, "reward": 0.546875, "reward_std": 0.29301512241363525, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1221, "tools/generated_tokens": 4672.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1027.9140625, "completions/mean_terminated_length": 968.9007568359375, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.20737694762647152, "epoch": 0.2082348179862398, "frac_reward_zero_std": 0.5, "grad_norm": 0.35590508580207825, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 503818165.0, "reward": 0.46875, "reward_std": 0.1896837055683136, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1222, "tools/generated_tokens": 4003.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1158.84765625, "completions/mean_terminated_length": 1079.3914794921875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.22455968149006367, "epoch": 0.20840522291094213, "frac_reward_zero_std": 0.375, "grad_norm": 0.4862659275531769, "learning_rate": 1e-06, "loss": 0.0016, "num_tokens": 504196398.0, "reward": 0.38671875, "reward_std": 0.26850375533103943, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1223, "tools/generated_tokens": 4414.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1066.61328125, "completions/mean_terminated_length": 1026.719482421875, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.16507458221167326, "epoch": 0.20857562783564446, "frac_reward_zero_std": 0.5625, "grad_norm": 0.22705848515033722, "learning_rate": 1e-06, "loss": 0.0044, "num_tokens": 504539579.0, "reward": 0.4375, "reward_std": 0.16713695228099823, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1224, "tools/generated_tokens": 3362.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1085.2421875, "completions/mean_terminated_length": 1008.0590209960938, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.1779328789561987, "epoch": 0.20874603276034678, "frac_reward_zero_std": 0.375, "grad_norm": 0.28504478931427, "learning_rate": 1e-06, "loss": 0.0054, "num_tokens": 504889257.0, "reward": 0.64453125, "reward_std": 0.2214793860912323, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1225, "tools/generated_tokens": 3813.2421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 969.0703125, "completions/mean_terminated_length": 929.757080078125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.17669182922691107, "epoch": 0.2089164376850491, "frac_reward_zero_std": 0.5, "grad_norm": 0.24836674332618713, "learning_rate": 1e-06, "loss": -0.0481, "num_tokens": 505213099.0, "reward": 0.75, "reward_std": 0.19013862311840057, "rewards/simpleverify_reward/mean": 0.75, "rewards/simpleverify_reward/std": 0.4338609278202057, "step": 1226, "tools/generated_tokens": 3481.06640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2032.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1123.38671875, "completions/mean_terminated_length": 1123.38671875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.17231587506830692, "epoch": 0.2090868426097514, "frac_reward_zero_std": 0.4375, "grad_norm": 0.29292505979537964, "learning_rate": 1e-06, "loss": -0.01, "num_tokens": 505580350.0, "reward": 0.53515625, "reward_std": 0.16846734285354614, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1227, "tools/generated_tokens": 3651.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1010.3515625, "completions/mean_terminated_length": 1002.1810913085938, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.17384162824600935, "epoch": 0.20925724753445374, "frac_reward_zero_std": 0.3125, "grad_norm": 0.42560380697250366, "learning_rate": 1e-06, "loss": -0.002, "num_tokens": 505911144.0, "reward": 0.66015625, "reward_std": 0.24943022429943085, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1228, "tools/generated_tokens": 3290.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 999.56640625, "completions/mean_terminated_length": 948.0040283203125, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.216273189522326, "epoch": 0.20942765245915607, "frac_reward_zero_std": 0.625, "grad_norm": 0.5432988405227661, "learning_rate": 1e-06, "loss": 0.0278, "num_tokens": 506255561.0, "reward": 0.5, "reward_std": 0.15975108742713928, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1229, "tools/generated_tokens": 4479.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.69921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1119.78125, "completions/mean_terminated_length": 1108.7747802734375, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.17066105920821428, "epoch": 0.2095980573838584, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3001425862312317, "learning_rate": 1e-06, "loss": 0.0341, "num_tokens": 506607793.0, "reward": 0.56640625, "reward_std": 0.18486405909061432, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1230, "tools/generated_tokens": 3423.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1171.15625, "completions/mean_terminated_length": 1153.6893310546875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.17302772123366594, "epoch": 0.20976846230856072, "frac_reward_zero_std": 0.4375, "grad_norm": 0.29006248712539673, "learning_rate": 1e-06, "loss": -0.0345, "num_tokens": 506971977.0, "reward": 0.5078125, "reward_std": 0.19036275148391724, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1231, "tools/generated_tokens": 3243.16015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1098.90625, "completions/mean_terminated_length": 1064.3238525390625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.1731760362163186, "epoch": 0.20993886723326305, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3600919544696808, "learning_rate": 1e-06, "loss": -0.0037, "num_tokens": 507321617.0, "reward": 0.40625, "reward_std": 0.20960845053195953, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1232, "tools/generated_tokens": 3626.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1172.046875, "completions/mean_terminated_length": 1085.579345703125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2018878674134612, "epoch": 0.21010927215796538, "frac_reward_zero_std": 0.375, "grad_norm": 0.32003867626190186, "learning_rate": 1e-06, "loss": 0.0442, "num_tokens": 507702189.0, "reward": 0.5703125, "reward_std": 0.2705788016319275, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1233, "tools/generated_tokens": 4324.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2000.0, "completions/mean_length": 1113.54296875, "completions/mean_terminated_length": 960.6363525390625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.21982169710099697, "epoch": 0.21027967708266768, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3235141634941101, "learning_rate": 1e-06, "loss": -0.0139, "num_tokens": 508077480.0, "reward": 0.44140625, "reward_std": 0.25284886360168457, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1234, "tools/generated_tokens": 5073.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1112.671875, "completions/mean_terminated_length": 1109.0040283203125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.187250723131001, "epoch": 0.21045008200737, "frac_reward_zero_std": 0.625, "grad_norm": 0.2581250071525574, "learning_rate": 1e-06, "loss": 0.0312, "num_tokens": 508424084.0, "reward": 0.58203125, "reward_std": 0.14138562977313995, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1235, "tools/generated_tokens": 3344.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.08984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1090.75, "completions/mean_terminated_length": 1000.752197265625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.1909032166004181, "epoch": 0.21062048693207233, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3132280707359314, "learning_rate": 1e-06, "loss": 0.0229, "num_tokens": 508791844.0, "reward": 0.53125, "reward_std": 0.22578103840351105, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1236, "tools/generated_tokens": 4378.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1045.28515625, "completions/mean_terminated_length": 1021.2200317382812, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.16267615463584661, "epoch": 0.21079089185677466, "frac_reward_zero_std": 0.5, "grad_norm": 0.231404647231102, "learning_rate": 1e-06, "loss": 0.0189, "num_tokens": 509140253.0, "reward": 0.4296875, "reward_std": 0.20630928874015808, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1237, "tools/generated_tokens": 3725.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 985.23828125, "completions/mean_terminated_length": 955.3613891601562, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.208086920902133, "epoch": 0.210961296781477, "frac_reward_zero_std": 0.375, "grad_norm": 0.3654596507549286, "learning_rate": 1e-06, "loss": -0.0057, "num_tokens": 509462858.0, "reward": 0.59375, "reward_std": 0.23088786005973816, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1238, "tools/generated_tokens": 3433.26171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1061.7265625, "completions/mean_terminated_length": 969.0000610351562, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.1821915190666914, "epoch": 0.21113170170617931, "frac_reward_zero_std": 0.625, "grad_norm": 0.27603158354759216, "learning_rate": 1e-06, "loss": 0.0263, "num_tokens": 509809380.0, "reward": 0.49609375, "reward_std": 0.14359626173973083, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1239, "tools/generated_tokens": 3845.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1065.8984375, "completions/mean_terminated_length": 1009.0825805664062, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.20180454198271036, "epoch": 0.21130210663088164, "frac_reward_zero_std": 0.375, "grad_norm": 0.348434716463089, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 510170058.0, "reward": 0.61328125, "reward_std": 0.2376519739627838, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1240, "tools/generated_tokens": 4409.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1039.90234375, "completions/mean_terminated_length": 1015.7080688476562, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.1852242313325405, "epoch": 0.21147251155558397, "frac_reward_zero_std": 0.4375, "grad_norm": 0.26168620586395264, "learning_rate": 1e-06, "loss": -0.0041, "num_tokens": 510512929.0, "reward": 0.59765625, "reward_std": 0.20673459768295288, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1241, "tools/generated_tokens": 3687.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1075.4921875, "completions/mean_terminated_length": 1052.152099609375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.1865016482770443, "epoch": 0.21164291648028627, "frac_reward_zero_std": 0.5, "grad_norm": 0.23254729807376862, "learning_rate": 1e-06, "loss": 0.0219, "num_tokens": 510862463.0, "reward": 0.58203125, "reward_std": 0.18727383017539978, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1242, "tools/generated_tokens": 3635.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1033.1328125, "completions/mean_terminated_length": 908.5, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.20272377599030733, "epoch": 0.2118133214049886, "frac_reward_zero_std": 0.5, "grad_norm": 0.3442402780056, "learning_rate": 1e-06, "loss": 0.0273, "num_tokens": 511205377.0, "reward": 0.67578125, "reward_std": 0.19860190153121948, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1243, "tools/generated_tokens": 3905.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1051.3515625, "completions/mean_terminated_length": 1015.0364379882812, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.18891155533492565, "epoch": 0.21198372632969092, "frac_reward_zero_std": 0.5625, "grad_norm": 0.32044512033462524, "learning_rate": 1e-06, "loss": 0.0325, "num_tokens": 511554875.0, "reward": 0.59375, "reward_std": 0.17333894968032837, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1244, "tools/generated_tokens": 4107.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1128.421875, "completions/mean_terminated_length": 1094.9150390625, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.18254025094211102, "epoch": 0.21215413125439325, "frac_reward_zero_std": 0.5, "grad_norm": 0.28156420588493347, "learning_rate": 1e-06, "loss": 0.0665, "num_tokens": 511911447.0, "reward": 0.62109375, "reward_std": 0.2054290771484375, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 1245, "tools/generated_tokens": 3600.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 1988.0, "completions/mean_length": 1091.671875, "completions/mean_terminated_length": 983.565185546875, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.201693763025105, "epoch": 0.21232453617909558, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28856873512268066, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 512268339.0, "reward": 0.63671875, "reward_std": 0.2356673926115036, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1246, "tools/generated_tokens": 4339.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1173.921875, "completions/mean_terminated_length": 1149.349365234375, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.1751686092466116, "epoch": 0.2124949411037979, "frac_reward_zero_std": 0.5, "grad_norm": 0.28663715720176697, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 512629311.0, "reward": 0.5390625, "reward_std": 0.1786910742521286, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1247, "tools/generated_tokens": 3653.921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1168.00390625, "completions/mean_terminated_length": 1117.094970703125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.18009694945067167, "epoch": 0.21266534602850024, "frac_reward_zero_std": 0.5, "grad_norm": 0.28416699171066284, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 513004288.0, "reward": 0.58203125, "reward_std": 0.212934672832489, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1248, "tools/generated_tokens": 3672.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 1998.0, "completions/mean_length": 1047.39453125, "completions/mean_terminated_length": 998.1843872070312, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.1795460507273674, "epoch": 0.21283575095320253, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2710454761981964, "learning_rate": 1e-06, "loss": -0.0068, "num_tokens": 513343461.0, "reward": 0.515625, "reward_std": 0.18199022114276886, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1249, "tools/generated_tokens": 3511.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1190.75390625, "completions/mean_terminated_length": 1085.482421875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.20419521816074848, "epoch": 0.21300615587790486, "frac_reward_zero_std": 0.5625, "grad_norm": 0.36411839723587036, "learning_rate": 1e-06, "loss": -0.0131, "num_tokens": 513727142.0, "reward": 0.57421875, "reward_std": 0.17770427465438843, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1250, "tools/generated_tokens": 4358.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1162.1015625, "completions/mean_terminated_length": 1129.8218994140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18734290450811386, "epoch": 0.2131765608026072, "frac_reward_zero_std": 0.5, "grad_norm": 0.28793346881866455, "learning_rate": 1e-06, "loss": -0.0008, "num_tokens": 514098704.0, "reward": 0.6171875, "reward_std": 0.19520533084869385, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1251, "tools/generated_tokens": 3698.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1188.5703125, "completions/mean_terminated_length": 1131.2750244140625, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.17425524443387985, "epoch": 0.21334696572730952, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2841739356517792, "learning_rate": 1e-06, "loss": 0.0286, "num_tokens": 514473618.0, "reward": 0.67578125, "reward_std": 0.23991455137729645, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1252, "tools/generated_tokens": 3788.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2019.0, "completions/mean_length": 987.88671875, "completions/mean_terminated_length": 907.7101440429688, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.19689524732530117, "epoch": 0.21351737065201185, "frac_reward_zero_std": 0.3125, "grad_norm": 0.32484301924705505, "learning_rate": 1e-06, "loss": -0.0225, "num_tokens": 514809109.0, "reward": 0.671875, "reward_std": 0.21478557586669922, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1253, "tools/generated_tokens": 3747.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 961.1328125, "completions/mean_terminated_length": 916.951171875, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.1871502296999097, "epoch": 0.21368777557671417, "frac_reward_zero_std": 0.5, "grad_norm": 0.2719271183013916, "learning_rate": 1e-06, "loss": -0.0104, "num_tokens": 515132279.0, "reward": 0.60546875, "reward_std": 0.17758671939373016, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1254, "tools/generated_tokens": 3905.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1118.84375, "completions/mean_terminated_length": 1096.5440673828125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.19255736097693443, "epoch": 0.2138581805014165, "frac_reward_zero_std": 0.5625, "grad_norm": 0.19443948566913605, "learning_rate": 1e-06, "loss": 0.0162, "num_tokens": 515499311.0, "reward": 0.5703125, "reward_std": 0.15284234285354614, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1255, "tools/generated_tokens": 3878.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1057.234375, "completions/mean_terminated_length": 1008.5081176757812, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.22242561168968678, "epoch": 0.21402858542611883, "frac_reward_zero_std": 0.375, "grad_norm": 0.43197181820869446, "learning_rate": 1e-06, "loss": -0.001, "num_tokens": 515851419.0, "reward": 0.5546875, "reward_std": 0.25702911615371704, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1256, "tools/generated_tokens": 4625.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2019.0, "completions/mean_length": 1158.31640625, "completions/mean_terminated_length": 1053.419189453125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.19480228889733553, "epoch": 0.21419899035082113, "frac_reward_zero_std": 0.4375, "grad_norm": 0.30144202709198, "learning_rate": 1e-06, "loss": 0.0344, "num_tokens": 516229740.0, "reward": 0.44140625, "reward_std": 0.23104894161224365, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1257, "tools/generated_tokens": 4422.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1081.109375, "completions/mean_terminated_length": 1069.644287109375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.1699511520564556, "epoch": 0.21436939527552346, "frac_reward_zero_std": 0.5625, "grad_norm": 0.24026639759540558, "learning_rate": 1e-06, "loss": 0.0113, "num_tokens": 516569000.0, "reward": 0.80859375, "reward_std": 0.18588893115520477, "rewards/simpleverify_reward/mean": 0.80859375, "rewards/simpleverify_reward/std": 0.39417871832847595, "step": 1258, "tools/generated_tokens": 3049.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1109.0078125, "completions/mean_terminated_length": 1020.7265625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.22308740857988596, "epoch": 0.21453980020022578, "frac_reward_zero_std": 0.5, "grad_norm": 0.43140721321105957, "learning_rate": 1e-06, "loss": -0.0246, "num_tokens": 516941354.0, "reward": 0.59765625, "reward_std": 0.19926717877388, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1259, "tools/generated_tokens": 5053.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1154.109375, "completions/mean_terminated_length": 1082.447265625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1889633135870099, "epoch": 0.2147102051249281, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3216223418712616, "learning_rate": 1e-06, "loss": 0.0267, "num_tokens": 517319526.0, "reward": 0.65625, "reward_std": 0.20403026044368744, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1260, "tools/generated_tokens": 4602.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1165.19921875, "completions/mean_terminated_length": 1154.7313232421875, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.1890478255227208, "epoch": 0.21488061004963044, "frac_reward_zero_std": 0.625, "grad_norm": 0.23615452647209167, "learning_rate": 1e-06, "loss": 0.0305, "num_tokens": 517692857.0, "reward": 0.66796875, "reward_std": 0.11752147227525711, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1261, "tools/generated_tokens": 3045.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1087.828125, "completions/mean_terminated_length": 1015.2101440429688, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.19847502931952477, "epoch": 0.21505101497433277, "frac_reward_zero_std": 0.625, "grad_norm": 0.27691033482551575, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 518048893.0, "reward": 0.47265625, "reward_std": 0.15663668513298035, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1262, "tools/generated_tokens": 3927.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1139.8515625, "completions/mean_terminated_length": 1079.308349609375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.21241279039531946, "epoch": 0.2152214198990351, "frac_reward_zero_std": 0.5625, "grad_norm": 0.27315887808799744, "learning_rate": 1e-06, "loss": 0.0361, "num_tokens": 518420199.0, "reward": 0.46875, "reward_std": 0.19394686818122864, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1263, "tools/generated_tokens": 4211.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1078.6796875, "completions/mean_terminated_length": 1067.185791015625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.17842696607112885, "epoch": 0.2153918248237374, "frac_reward_zero_std": 0.5625, "grad_norm": 0.43259936571121216, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 518761749.0, "reward": 0.63671875, "reward_std": 0.15646496415138245, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1264, "tools/generated_tokens": 3030.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1076.01171875, "completions/mean_terminated_length": 1040.59521484375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.20656804367899895, "epoch": 0.21556222974843972, "frac_reward_zero_std": 0.375, "grad_norm": 0.28473782539367676, "learning_rate": 1e-06, "loss": -0.0032, "num_tokens": 519117368.0, "reward": 0.67578125, "reward_std": 0.24237701296806335, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1265, "tools/generated_tokens": 3748.015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1092.7890625, "completions/mean_terminated_length": 1085.2677001953125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.19266426283866167, "epoch": 0.21573263467314205, "frac_reward_zero_std": 0.4375, "grad_norm": 0.26350802183151245, "learning_rate": 1e-06, "loss": 0.0185, "num_tokens": 519469954.0, "reward": 0.8046875, "reward_std": 0.18375971913337708, "rewards/simpleverify_reward/mean": 0.8046875, "rewards/simpleverify_reward/std": 0.39721766114234924, "step": 1266, "tools/generated_tokens": 3516.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1236.01171875, "completions/mean_terminated_length": 1111.6531982421875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.2031150683760643, "epoch": 0.21590303959784438, "frac_reward_zero_std": 0.4375, "grad_norm": 0.36398014426231384, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 519861317.0, "reward": 0.55078125, "reward_std": 0.20200955867767334, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1267, "tools/generated_tokens": 4652.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1117.3046875, "completions/mean_terminated_length": 1025.4334716796875, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.21470707468688488, "epoch": 0.2160734445225467, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2354200780391693, "learning_rate": 1e-06, "loss": 0.0094, "num_tokens": 520231219.0, "reward": 0.46484375, "reward_std": 0.18207119405269623, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1268, "tools/generated_tokens": 4661.3046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1153.4921875, "completions/mean_terminated_length": 1089.8660888671875, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.18564417958259583, "epoch": 0.21624384944724903, "frac_reward_zero_std": 0.4375, "grad_norm": 0.27607589960098267, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 520596657.0, "reward": 0.48828125, "reward_std": 0.215663880109787, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1269, "tools/generated_tokens": 3753.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1090.3125, "completions/mean_terminated_length": 1000.2735595703125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.16396806854754686, "epoch": 0.21641425437195136, "frac_reward_zero_std": 0.4375, "grad_norm": 1.7126364707946777, "learning_rate": 1e-06, "loss": 0.0306, "num_tokens": 520963169.0, "reward": 0.70703125, "reward_std": 0.20091910660266876, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 1270, "tools/generated_tokens": 3618.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1138.65234375, "completions/mean_terminated_length": 1048.888427734375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.21545883920043707, "epoch": 0.2165846592966537, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2614912986755371, "learning_rate": 1e-06, "loss": -0.0207, "num_tokens": 521339432.0, "reward": 0.44921875, "reward_std": 0.23292499780654907, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1271, "tools/generated_tokens": 4650.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2043.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1112.86328125, "completions/mean_terminated_length": 1112.86328125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.1561153121292591, "epoch": 0.216755064221356, "frac_reward_zero_std": 0.5, "grad_norm": 0.28968772292137146, "learning_rate": 1e-06, "loss": -0.0086, "num_tokens": 521690245.0, "reward": 0.734375, "reward_std": 0.1687908172607422, "rewards/simpleverify_reward/mean": 0.734375, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 1272, "tools/generated_tokens": 2816.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.83203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1054.30078125, "completions/mean_terminated_length": 902.1126098632812, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.2100852783769369, "epoch": 0.21692546914605831, "frac_reward_zero_std": 0.375, "grad_norm": 0.34271353483200073, "learning_rate": 1e-06, "loss": 0.0321, "num_tokens": 522042386.0, "reward": 0.5234375, "reward_std": 0.2575877904891968, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1273, "tools/generated_tokens": 4366.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1002.54296875, "completions/mean_terminated_length": 913.9449462890625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.1744906473904848, "epoch": 0.21709587407076064, "frac_reward_zero_std": 0.4375, "grad_norm": 0.36521026492118835, "learning_rate": 1e-06, "loss": 0.0402, "num_tokens": 522379453.0, "reward": 0.54296875, "reward_std": 0.2576804757118225, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1274, "tools/generated_tokens": 4322.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1111.9296875, "completions/mean_terminated_length": 1069.9019775390625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.1923376303166151, "epoch": 0.21726627899546297, "frac_reward_zero_std": 0.4375, "grad_norm": 0.30232682824134827, "learning_rate": 1e-06, "loss": -0.0468, "num_tokens": 522732283.0, "reward": 0.6796875, "reward_std": 0.22156035900115967, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1275, "tools/generated_tokens": 3751.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 941.63671875, "completions/mean_terminated_length": 887.225341796875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.1668767612427473, "epoch": 0.2174366839201653, "frac_reward_zero_std": 0.5625, "grad_norm": 0.29059362411499023, "learning_rate": 1e-06, "loss": 0.0257, "num_tokens": 523046430.0, "reward": 0.68359375, "reward_std": 0.15789085626602173, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1276, "tools/generated_tokens": 3461.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1230.08984375, "completions/mean_terminated_length": 1182.772705078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.18885768298059702, "epoch": 0.21760708884486762, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2807115316390991, "learning_rate": 1e-06, "loss": 0.079, "num_tokens": 523434837.0, "reward": 0.65625, "reward_std": 0.23914283514022827, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1277, "tools/generated_tokens": 3942.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1015.76953125, "completions/mean_terminated_length": 978.1578979492188, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.19331876002252102, "epoch": 0.21777749376956995, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2733987867832184, "learning_rate": 1e-06, "loss": 0.0007, "num_tokens": 523777434.0, "reward": 0.6796875, "reward_std": 0.18119096755981445, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1278, "tools/generated_tokens": 3887.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1143.98828125, "completions/mean_terminated_length": 1118.57421875, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.1707342453300953, "epoch": 0.21794789869427225, "frac_reward_zero_std": 0.625, "grad_norm": 0.2875167727470398, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 524132839.0, "reward": 0.67578125, "reward_std": 0.16141413152217865, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1279, "tools/generated_tokens": 3087.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.94921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1127.22265625, "completions/mean_terminated_length": 1077.962890625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.1849861079826951, "epoch": 0.21811830361897458, "frac_reward_zero_std": 0.5625, "grad_norm": 0.25774428248405457, "learning_rate": 1e-06, "loss": 0.0077, "num_tokens": 524497312.0, "reward": 0.50390625, "reward_std": 0.162959486246109, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1280, "tools/generated_tokens": 4047.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.42578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2009.0, "completions/mean_length": 1118.12890625, "completions/mean_terminated_length": 1043.582275390625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.18623533844947815, "epoch": 0.2182887085436769, "frac_reward_zero_std": 0.5625, "grad_norm": 0.6831443905830383, "learning_rate": 1e-06, "loss": 0.0383, "num_tokens": 524858081.0, "reward": 0.5859375, "reward_std": 0.16713695228099823, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1281, "tools/generated_tokens": 3974.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1022.9296875, "completions/mean_terminated_length": 976.9060668945312, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.18899204675108194, "epoch": 0.21845911346837923, "frac_reward_zero_std": 0.375, "grad_norm": 0.3431737422943115, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 525202063.0, "reward": 0.5390625, "reward_std": 0.24541130661964417, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1282, "tools/generated_tokens": 4526.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1054.83984375, "completions/mean_terminated_length": 1014.4674682617188, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.15869122836738825, "epoch": 0.21862951839308156, "frac_reward_zero_std": 0.375, "grad_norm": 0.3238829970359802, "learning_rate": 1e-06, "loss": 0.0248, "num_tokens": 525551590.0, "reward": 0.5703125, "reward_std": 0.22644630074501038, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1283, "tools/generated_tokens": 3710.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1130.03515625, "completions/mean_terminated_length": 1052.2415771484375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.18005719128996134, "epoch": 0.2187999233177839, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2748975157737732, "learning_rate": 1e-06, "loss": 0.0057, "num_tokens": 525919615.0, "reward": 0.5546875, "reward_std": 0.22270601987838745, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1284, "tools/generated_tokens": 4362.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1105.80859375, "completions/mean_terminated_length": 1094.6363525390625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.18099006544798613, "epoch": 0.21897032824248622, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3657585382461548, "learning_rate": 1e-06, "loss": 0.0477, "num_tokens": 526268814.0, "reward": 0.5859375, "reward_std": 0.2665986716747284, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1285, "tools/generated_tokens": 3553.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1050.68359375, "completions/mean_terminated_length": 966.165283203125, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.1923149712383747, "epoch": 0.21914073316718855, "frac_reward_zero_std": 0.25, "grad_norm": 0.44675296545028687, "learning_rate": 1e-06, "loss": -0.0043, "num_tokens": 526620029.0, "reward": 0.4296875, "reward_std": 0.2666005790233612, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1286, "tools/generated_tokens": 4290.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1172.4453125, "completions/mean_terminated_length": 1140.54248046875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.1674215542152524, "epoch": 0.21931113809189084, "frac_reward_zero_std": 0.25, "grad_norm": 0.34224122762680054, "learning_rate": 1e-06, "loss": -0.0028, "num_tokens": 526987487.0, "reward": 0.69140625, "reward_std": 0.2468073070049286, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1287, "tools/generated_tokens": 3188.4453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1156.13671875, "completions/mean_terminated_length": 1119.882080078125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.18488428834825754, "epoch": 0.21948154301659317, "frac_reward_zero_std": 0.375, "grad_norm": 0.34940001368522644, "learning_rate": 1e-06, "loss": 0.018, "num_tokens": 527353826.0, "reward": 0.6796875, "reward_std": 0.2620202898979187, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1288, "tools/generated_tokens": 3772.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1120.78515625, "completions/mean_terminated_length": 1058.970947265625, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.21201751567423344, "epoch": 0.2196519479412955, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2808263599872589, "learning_rate": 1e-06, "loss": 0.0218, "num_tokens": 527729771.0, "reward": 0.47265625, "reward_std": 0.19311904907226562, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1289, "tools/generated_tokens": 5264.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1248.49609375, "completions/mean_terminated_length": 1169.5750732421875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.18602818436920643, "epoch": 0.21982235286599783, "frac_reward_zero_std": 0.5625, "grad_norm": 0.31078609824180603, "learning_rate": 1e-06, "loss": 0.0261, "num_tokens": 528128602.0, "reward": 0.48828125, "reward_std": 0.1917063295841217, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1290, "tools/generated_tokens": 4544.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1259.5625, "completions/mean_terminated_length": 1138.8243408203125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.1548333838582039, "epoch": 0.21999275779070016, "frac_reward_zero_std": 0.6875, "grad_norm": 0.22505289316177368, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 528524554.0, "reward": 0.51953125, "reward_std": 0.1257193237543106, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1291, "tools/generated_tokens": 4339.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.50390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1203.625, "completions/mean_terminated_length": 1172.8582763671875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.15039650443941355, "epoch": 0.22016316271540248, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3051331639289856, "learning_rate": 1e-06, "loss": 0.0414, "num_tokens": 528900602.0, "reward": 0.6171875, "reward_std": 0.2665349841117859, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1292, "tools/generated_tokens": 3555.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1245.47265625, "completions/mean_terminated_length": 1142.9471435546875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.18847098108381033, "epoch": 0.2203335676401048, "frac_reward_zero_std": 0.5, "grad_norm": 0.314424991607666, "learning_rate": 1e-06, "loss": 0.0106, "num_tokens": 529289091.0, "reward": 0.39453125, "reward_std": 0.18584169447422028, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1293, "tools/generated_tokens": 4229.46875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1055.703125, "completions/mean_terminated_length": 1015.3658447265625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.15517426189035177, "epoch": 0.2205039725648071, "frac_reward_zero_std": 0.625, "grad_norm": 0.2526600658893585, "learning_rate": 1e-06, "loss": -0.0042, "num_tokens": 529629863.0, "reward": 0.5, "reward_std": 0.1290597915649414, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1294, "tools/generated_tokens": 3471.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1239.6953125, "completions/mean_terminated_length": 1140.4385986328125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.18641173094511032, "epoch": 0.22067437748950944, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3590131998062134, "learning_rate": 1e-06, "loss": 0.0215, "num_tokens": 530021609.0, "reward": 0.53515625, "reward_std": 0.17770425975322723, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1295, "tools/generated_tokens": 4359.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1080.19140625, "completions/mean_terminated_length": 1036.73876953125, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.14707045629620552, "epoch": 0.22084478241421177, "frac_reward_zero_std": 0.5, "grad_norm": 0.31944042444229126, "learning_rate": 1e-06, "loss": 0.0273, "num_tokens": 530366090.0, "reward": 0.578125, "reward_std": 0.19816282391548157, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1296, "tools/generated_tokens": 3728.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1057.8515625, "completions/mean_terminated_length": 1004.880615234375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.18700439017266035, "epoch": 0.2210151873389141, "frac_reward_zero_std": 0.375, "grad_norm": 0.37505123019218445, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 530714852.0, "reward": 0.48046875, "reward_std": 0.246913880109787, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1297, "tools/generated_tokens": 4345.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1213.26171875, "completions/mean_terminated_length": 1193.22802734375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.1574186086654663, "epoch": 0.22118559226361642, "frac_reward_zero_std": 0.75, "grad_norm": 0.25454750657081604, "learning_rate": 1e-06, "loss": 0.002, "num_tokens": 531090167.0, "reward": 0.6953125, "reward_std": 0.10981409251689911, "rewards/simpleverify_reward/mean": 0.6953125, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 1298, "tools/generated_tokens": 3669.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1252.90625, "completions/mean_terminated_length": 1087.8868408203125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.17490413505584002, "epoch": 0.22135599718831875, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3861309885978699, "learning_rate": 1e-06, "loss": 0.0013, "num_tokens": 531491855.0, "reward": 0.51953125, "reward_std": 0.22808240354061127, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1299, "tools/generated_tokens": 4988.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1112.82421875, "completions/mean_terminated_length": 1050.479248046875, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.18131343368440866, "epoch": 0.22152640211302108, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3058548867702484, "learning_rate": 1e-06, "loss": 0.0024, "num_tokens": 531847186.0, "reward": 0.58984375, "reward_std": 0.1571033000946045, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1300, "tools/generated_tokens": 3448.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1085.125, "completions/mean_terminated_length": 1058.05615234375, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.16897037532180548, "epoch": 0.2216968070377234, "frac_reward_zero_std": 0.6875, "grad_norm": 0.22796973586082458, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 532222386.0, "reward": 0.28515625, "reward_std": 0.1284485161304474, "rewards/simpleverify_reward/mean": 0.28515625, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 1301, "tools/generated_tokens": 3869.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1029.5078125, "completions/mean_terminated_length": 1000.87548828125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1502059819176793, "epoch": 0.2218672119624257, "frac_reward_zero_std": 0.75, "grad_norm": 0.20146328210830688, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 532563572.0, "reward": 0.72265625, "reward_std": 0.09208697080612183, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1302, "tools/generated_tokens": 3365.50390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1077.1484375, "completions/mean_terminated_length": 1020.9833984375, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.16898761317133904, "epoch": 0.22203761688712803, "frac_reward_zero_std": 0.625, "grad_norm": 0.3000277578830719, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 532923162.0, "reward": 0.5703125, "reward_std": 0.14294016361236572, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1303, "tools/generated_tokens": 4381.14453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.61328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1167.09375, "completions/mean_terminated_length": 1022.9454345703125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.17193921096622944, "epoch": 0.22220802181183036, "frac_reward_zero_std": 0.5, "grad_norm": 0.26873812079429626, "learning_rate": 1e-06, "loss": 0.0216, "num_tokens": 533306594.0, "reward": 0.3515625, "reward_std": 0.1979367733001709, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1304, "tools/generated_tokens": 4855.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2040.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1183.13671875, "completions/mean_terminated_length": 1183.13671875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.15278471447527409, "epoch": 0.22237842673653269, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2617805302143097, "learning_rate": 1e-06, "loss": -0.0127, "num_tokens": 533682213.0, "reward": 0.79296875, "reward_std": 0.19005721807479858, "rewards/simpleverify_reward/mean": 0.79296875, "rewards/simpleverify_reward/std": 0.40597182512283325, "step": 1305, "tools/generated_tokens": 3271.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1182.4375, "completions/mean_terminated_length": 1036.200927734375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.15132269263267517, "epoch": 0.222548831661235, "frac_reward_zero_std": 0.5, "grad_norm": 0.27412426471710205, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 534056933.0, "reward": 0.640625, "reward_std": 0.237278014421463, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1306, "tools/generated_tokens": 4478.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1091.0546875, "completions/mean_terminated_length": 1064.152587890625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.14944705460220575, "epoch": 0.22271923658593734, "frac_reward_zero_std": 0.625, "grad_norm": 0.2172091007232666, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 534406739.0, "reward": 0.52734375, "reward_std": 0.1598844826221466, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1307, "tools/generated_tokens": 3627.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1179.203125, "completions/mean_terminated_length": 1105.5762939453125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1890433980152011, "epoch": 0.22288964151063967, "frac_reward_zero_std": 0.4375, "grad_norm": 0.280519962310791, "learning_rate": 1e-06, "loss": 0.0238, "num_tokens": 534786199.0, "reward": 0.484375, "reward_std": 0.22532612085342407, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1308, "tools/generated_tokens": 4259.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.50390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1157.70703125, "completions/mean_terminated_length": 1094.3807373046875, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.1508091939613223, "epoch": 0.22306004643534197, "frac_reward_zero_std": 0.6875, "grad_norm": 0.23531325161457062, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 535159228.0, "reward": 0.609375, "reward_std": 0.11431500315666199, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1309, "tools/generated_tokens": 3869.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1041.6640625, "completions/mean_terminated_length": 1017.5120239257812, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.18083287589251995, "epoch": 0.2232304513600443, "frac_reward_zero_std": 0.3125, "grad_norm": 0.38047516345977783, "learning_rate": 1e-06, "loss": -0.0198, "num_tokens": 535505782.0, "reward": 0.64453125, "reward_std": 0.27067145705223083, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1310, "tools/generated_tokens": 3705.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1186.515625, "completions/mean_terminated_length": 1121.3656005859375, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.17253058031201363, "epoch": 0.22340085628474662, "frac_reward_zero_std": 0.5, "grad_norm": 0.2898007035255432, "learning_rate": 1e-06, "loss": -0.0068, "num_tokens": 535893994.0, "reward": 0.453125, "reward_std": 0.19401052594184875, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1311, "tools/generated_tokens": 4778.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.75390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1282.2109375, "completions/mean_terminated_length": 1188.1666259765625, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.1620686212554574, "epoch": 0.22357126120944895, "frac_reward_zero_std": 0.5, "grad_norm": 0.255595326423645, "learning_rate": 1e-06, "loss": 0.0333, "num_tokens": 536291984.0, "reward": 0.51171875, "reward_std": 0.2142539918422699, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1312, "tools/generated_tokens": 4330.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1133.6875, "completions/mean_terminated_length": 1072.7333984375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.19369827955961227, "epoch": 0.22374166613415128, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3146739900112152, "learning_rate": 1e-06, "loss": 0.0327, "num_tokens": 536667888.0, "reward": 0.35546875, "reward_std": 0.19938471913337708, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1313, "tools/generated_tokens": 5269.69140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1196.1484375, "completions/mean_terminated_length": 1074.4554443359375, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.1728651002049446, "epoch": 0.2239120710588536, "frac_reward_zero_std": 0.25, "grad_norm": 69.7646713256836, "learning_rate": 1e-06, "loss": 0.0518, "num_tokens": 537053814.0, "reward": 0.58203125, "reward_std": 0.29984626173973083, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1314, "tools/generated_tokens": 4460.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1111.07421875, "completions/mean_terminated_length": 1060.9505615234375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.16822955291718245, "epoch": 0.22408247598355593, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3103754222393036, "learning_rate": 1e-06, "loss": -0.019, "num_tokens": 537412729.0, "reward": 0.44140625, "reward_std": 0.23758356273174286, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1315, "tools/generated_tokens": 4311.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1024.453125, "completions/mean_terminated_length": 991.4354248046875, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.1744611831381917, "epoch": 0.22425288090825826, "frac_reward_zero_std": 0.375, "grad_norm": 0.2998579144477844, "learning_rate": 1e-06, "loss": 0.0635, "num_tokens": 537763581.0, "reward": 0.51953125, "reward_std": 0.2611019015312195, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1316, "tools/generated_tokens": 4320.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 965.58984375, "completions/mean_terminated_length": 944.0278930664062, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.1851312117651105, "epoch": 0.22442328583296056, "frac_reward_zero_std": 0.5, "grad_norm": 0.33820584416389465, "learning_rate": 1e-06, "loss": 0.026, "num_tokens": 538098356.0, "reward": 0.46875, "reward_std": 0.16406384110450745, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1317, "tools/generated_tokens": 3901.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2014.0, "completions/mean_length": 1147.78515625, "completions/mean_terminated_length": 1041.6463623046875, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.16480957716703415, "epoch": 0.2245936907576629, "frac_reward_zero_std": 0.6875, "grad_norm": 0.22834157943725586, "learning_rate": 1e-06, "loss": -0.0182, "num_tokens": 538469405.0, "reward": 0.51953125, "reward_std": 0.11234625428915024, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1318, "tools/generated_tokens": 4067.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.42578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1076.43359375, "completions/mean_terminated_length": 1007.3263549804688, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.1489297365769744, "epoch": 0.22476409568236522, "frac_reward_zero_std": 0.5625, "grad_norm": 0.26802268624305725, "learning_rate": 1e-06, "loss": -0.0201, "num_tokens": 538817180.0, "reward": 0.57421875, "reward_std": 0.15458697080612183, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1319, "tools/generated_tokens": 3356.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1013.08203125, "completions/mean_terminated_length": 957.7160034179688, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.19144816976040602, "epoch": 0.22493450060706754, "frac_reward_zero_std": 0.375, "grad_norm": 0.5645748376846313, "learning_rate": 1e-06, "loss": 0.0606, "num_tokens": 539165057.0, "reward": 0.46484375, "reward_std": 0.23217815160751343, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1320, "tools/generated_tokens": 5077.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1022.1328125, "completions/mean_terminated_length": 997.5120239257812, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.15609671734273434, "epoch": 0.22510490553176987, "frac_reward_zero_std": 0.4375, "grad_norm": 0.2892788350582123, "learning_rate": 1e-06, "loss": -0.0191, "num_tokens": 539507923.0, "reward": 0.58984375, "reward_std": 0.19916057586669922, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1321, "tools/generated_tokens": 3518.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1138.2734375, "completions/mean_terminated_length": 1008.3125610351562, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.18075208831578493, "epoch": 0.2252753104564722, "frac_reward_zero_std": 0.4375, "grad_norm": 0.38152769207954407, "learning_rate": 1e-06, "loss": -0.0016, "num_tokens": 539884265.0, "reward": 0.61328125, "reward_std": 0.2163580358028412, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1322, "tools/generated_tokens": 4450.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1049.3671875, "completions/mean_terminated_length": 1000.2540283203125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.16696893703192472, "epoch": 0.22544571538117453, "frac_reward_zero_std": 0.4375, "grad_norm": 0.33197301626205444, "learning_rate": 1e-06, "loss": -0.0082, "num_tokens": 540232295.0, "reward": 0.5390625, "reward_std": 0.18409234285354614, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1323, "tools/generated_tokens": 4521.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2011.0, "completions/mean_length": 983.9453125, "completions/mean_terminated_length": 940.6910400390625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.17016990762203932, "epoch": 0.22561612030587683, "frac_reward_zero_std": 0.3125, "grad_norm": 0.2880268394947052, "learning_rate": 1e-06, "loss": 0.0032, "num_tokens": 540555337.0, "reward": 0.58203125, "reward_std": 0.22271710634231567, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1324, "tools/generated_tokens": 3519.9453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1041.546875, "completions/mean_terminated_length": 1037.60009765625, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.1773486640304327, "epoch": 0.22578652523057915, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28657054901123047, "learning_rate": 1e-06, "loss": -0.0039, "num_tokens": 540895077.0, "reward": 0.63671875, "reward_std": 0.20410975813865662, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1325, "tools/generated_tokens": 3497.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1128.24609375, "completions/mean_terminated_length": 1071.0, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.13714859075844288, "epoch": 0.22595693015528148, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2882538139820099, "learning_rate": 1e-06, "loss": 0.0036, "num_tokens": 541243556.0, "reward": 0.73046875, "reward_std": 0.16471801698207855, "rewards/simpleverify_reward/mean": 0.73046875, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 1326, "tools/generated_tokens": 3096.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1053.56640625, "completions/mean_terminated_length": 1021.4878540039062, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.15736063476651907, "epoch": 0.2261273350799838, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2666698098182678, "learning_rate": 1e-06, "loss": 0.0245, "num_tokens": 541580005.0, "reward": 0.65234375, "reward_std": 0.11046825349330902, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1327, "tools/generated_tokens": 3445.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.16796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2040.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1169.90625, "completions/mean_terminated_length": 1169.90625, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.17227158416062593, "epoch": 0.22629774000468614, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2718101441860199, "learning_rate": 1e-06, "loss": -0.0086, "num_tokens": 541950605.0, "reward": 0.58984375, "reward_std": 0.12213994562625885, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1328, "tools/generated_tokens": 3745.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1061.328125, "completions/mean_terminated_length": 995.550048828125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.1359957456588745, "epoch": 0.22646814492938847, "frac_reward_zero_std": 0.75, "grad_norm": 0.1845640391111374, "learning_rate": 1e-06, "loss": -0.0012, "num_tokens": 542298497.0, "reward": 0.55859375, "reward_std": 0.10489008575677872, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1329, "tools/generated_tokens": 3781.33203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1025.70703125, "completions/mean_terminated_length": 996.9678344726562, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.17172004468739033, "epoch": 0.2266385498540908, "frac_reward_zero_std": 0.4375, "grad_norm": 0.37054407596588135, "learning_rate": 1e-06, "loss": 0.0236, "num_tokens": 542647558.0, "reward": 0.58203125, "reward_std": 0.21072597801685333, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1330, "tools/generated_tokens": 4265.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 992.36328125, "completions/mean_terminated_length": 967.028076171875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.16265062615275383, "epoch": 0.22680895477879312, "frac_reward_zero_std": 0.4375, "grad_norm": 0.35218170285224915, "learning_rate": 1e-06, "loss": -0.0037, "num_tokens": 542984867.0, "reward": 0.60546875, "reward_std": 0.20194396376609802, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1331, "tools/generated_tokens": 3920.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 986.1171875, "completions/mean_terminated_length": 956.2650146484375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.1809298498556018, "epoch": 0.22697935970349542, "frac_reward_zero_std": 0.5, "grad_norm": 0.4008549153804779, "learning_rate": 1e-06, "loss": 0.0277, "num_tokens": 543313121.0, "reward": 0.44140625, "reward_std": 0.1701192855834961, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1332, "tools/generated_tokens": 4050.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1078.66796875, "completions/mean_terminated_length": 1063.2818603515625, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.14357508718967438, "epoch": 0.22714976462819775, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3023888170719147, "learning_rate": 1e-06, "loss": 0.0136, "num_tokens": 543658284.0, "reward": 0.671875, "reward_std": 0.15513263642787933, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1333, "tools/generated_tokens": 3822.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1012.0078125, "completions/mean_terminated_length": 1003.8504028320312, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.15268967393785715, "epoch": 0.22732016955290008, "frac_reward_zero_std": 0.6875, "grad_norm": 0.27028632164001465, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 543981598.0, "reward": 0.78125, "reward_std": 0.11282352358102798, "rewards/simpleverify_reward/mean": 0.78125, "rewards/simpleverify_reward/std": 0.41420844197273254, "step": 1334, "tools/generated_tokens": 3124.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.03125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1160.4140625, "completions/mean_terminated_length": 1142.733154296875, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.12800501380115747, "epoch": 0.2274905744776024, "frac_reward_zero_std": 0.625, "grad_norm": 0.328451007604599, "learning_rate": 1e-06, "loss": 0.0185, "num_tokens": 544332904.0, "reward": 0.68359375, "reward_std": 0.14963588118553162, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1335, "tools/generated_tokens": 3088.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.94140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1031.3203125, "completions/mean_terminated_length": 1027.3333740234375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.17485059145838022, "epoch": 0.22766097940230473, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5302588939666748, "learning_rate": 1e-06, "loss": -0.0537, "num_tokens": 544682410.0, "reward": 0.40625, "reward_std": 0.27705955505371094, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1336, "tools/generated_tokens": 4535.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1130.71875, "completions/mean_terminated_length": 1116.1588134765625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.14649152848869562, "epoch": 0.22783138432700706, "frac_reward_zero_std": 0.375, "grad_norm": 0.362393319606781, "learning_rate": 1e-06, "loss": 0.0009, "num_tokens": 545030914.0, "reward": 0.49609375, "reward_std": 0.24407809972763062, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1337, "tools/generated_tokens": 3258.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1023.7265625, "completions/mean_terminated_length": 1011.5810546875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.14802028611302376, "epoch": 0.22800178925170939, "frac_reward_zero_std": 0.5, "grad_norm": 0.38212326169013977, "learning_rate": 1e-06, "loss": 0.0241, "num_tokens": 545371516.0, "reward": 0.72265625, "reward_std": 0.14029237627983093, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1338, "tools/generated_tokens": 3231.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1073.2421875, "completions/mean_terminated_length": 1003.9078979492188, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1445963028818369, "epoch": 0.22817219417641169, "frac_reward_zero_std": 0.625, "grad_norm": 0.2530478537082672, "learning_rate": 1e-06, "loss": 0.0063, "num_tokens": 545718554.0, "reward": 0.4765625, "reward_std": 0.10519562661647797, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1339, "tools/generated_tokens": 4201.2421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1091.70703125, "completions/mean_terminated_length": 1032.186767578125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.1454448401927948, "epoch": 0.228342599101114, "frac_reward_zero_std": 0.625, "grad_norm": 0.29997122287750244, "learning_rate": 1e-06, "loss": 0.0166, "num_tokens": 546072143.0, "reward": 0.6015625, "reward_std": 0.15029004216194153, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1340, "tools/generated_tokens": 4099.7109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2014.0, "completions/mean_length": 1048.00390625, "completions/mean_terminated_length": 958.6425170898438, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1514856368303299, "epoch": 0.22851300402581634, "frac_reward_zero_std": 0.4375, "grad_norm": 0.35856449604034424, "learning_rate": 1e-06, "loss": 0.0225, "num_tokens": 546422560.0, "reward": 0.546875, "reward_std": 0.21723634004592896, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1341, "tools/generated_tokens": 4168.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1115.875, "completions/mean_terminated_length": 1112.2197265625, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.14187026489526033, "epoch": 0.22868340895051867, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3230423331260681, "learning_rate": 1e-06, "loss": -0.0282, "num_tokens": 546773360.0, "reward": 0.6171875, "reward_std": 0.20476585626602173, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1342, "tools/generated_tokens": 3563.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1183.46875, "completions/mean_terminated_length": 1121.974853515625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.1428342815488577, "epoch": 0.228853813875221, "frac_reward_zero_std": 0.5, "grad_norm": 0.30634552240371704, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 547155224.0, "reward": 0.48828125, "reward_std": 0.16494406759738922, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1343, "tools/generated_tokens": 4311.46875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1053.17578125, "completions/mean_terminated_length": 1041.3795166015625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.1509106820449233, "epoch": 0.22902421879992332, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3548799455165863, "learning_rate": 1e-06, "loss": -0.0027, "num_tokens": 547504901.0, "reward": 0.69140625, "reward_std": 0.23770393431186676, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1344, "tools/generated_tokens": 3581.171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1129.47265625, "completions/mean_terminated_length": 1107.4281005859375, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.1773249702528119, "epoch": 0.22919462372462565, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2641788125038147, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 547858014.0, "reward": 0.5703125, "reward_std": 0.17191587388515472, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1345, "tools/generated_tokens": 3737.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1083.3125, "completions/mean_terminated_length": 1075.716552734375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.15080509707331657, "epoch": 0.22936502864932798, "frac_reward_zero_std": 0.625, "grad_norm": 0.2641783654689789, "learning_rate": 1e-06, "loss": 0.0077, "num_tokens": 548194542.0, "reward": 0.55859375, "reward_std": 0.14423459768295288, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1346, "tools/generated_tokens": 2971.3125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1106.7734375, "completions/mean_terminated_length": 1060.4835205078125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.1563136139884591, "epoch": 0.22953543357403028, "frac_reward_zero_std": 0.4375, "grad_norm": 0.42173317074775696, "learning_rate": 1e-06, "loss": 0.0355, "num_tokens": 548553748.0, "reward": 0.54296875, "reward_std": 0.2148810774087906, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1347, "tools/generated_tokens": 3898.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1018.9140625, "completions/mean_terminated_length": 981.4170532226562, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.17680420354008675, "epoch": 0.2297058384987326, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3469582498073578, "learning_rate": 1e-06, "loss": 0.0422, "num_tokens": 548903774.0, "reward": 0.5859375, "reward_std": 0.1617356687784195, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1348, "tools/generated_tokens": 4274.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1113.9296875, "completions/mean_terminated_length": 1055.7926025390625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.17317654378712177, "epoch": 0.22987624342343493, "frac_reward_zero_std": 0.3125, "grad_norm": 0.7226481437683105, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 549273916.0, "reward": 0.453125, "reward_std": 0.2499074935913086, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1349, "tools/generated_tokens": 4713.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1141.99609375, "completions/mean_terminated_length": 1101.318359375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.14713975880295038, "epoch": 0.23004664834813726, "frac_reward_zero_std": 0.4375, "grad_norm": 0.31657806038856506, "learning_rate": 1e-06, "loss": -0.0098, "num_tokens": 549638731.0, "reward": 0.76953125, "reward_std": 0.20707818865776062, "rewards/simpleverify_reward/mean": 0.76953125, "rewards/simpleverify_reward/std": 0.4219578504562378, "step": 1350, "tools/generated_tokens": 3893.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1118.984375, "completions/mean_terminated_length": 1027.2789306640625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.15826607309281826, "epoch": 0.2302170532728396, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3316590189933777, "learning_rate": 1e-06, "loss": 0.0358, "num_tokens": 550011703.0, "reward": 0.52734375, "reward_std": 0.21578142046928406, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1351, "tools/generated_tokens": 4510.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1118.5625, "completions/mean_terminated_length": 1068.8394775390625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.1821862170472741, "epoch": 0.23038745819754192, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3441648483276367, "learning_rate": 1e-06, "loss": -0.0045, "num_tokens": 550377399.0, "reward": 0.4765625, "reward_std": 0.23140643537044525, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1352, "tools/generated_tokens": 4366.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1109.53125, "completions/mean_terminated_length": 1034.2952880859375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.1856785686686635, "epoch": 0.23055786312224424, "frac_reward_zero_std": 0.625, "grad_norm": 0.25652840733528137, "learning_rate": 1e-06, "loss": 0.0126, "num_tokens": 550745743.0, "reward": 0.40234375, "reward_std": 0.14501741528511047, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1353, "tools/generated_tokens": 4373.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1289.5703125, "completions/mean_terminated_length": 1149.120361328125, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.15360585507005453, "epoch": 0.23072826804694654, "frac_reward_zero_std": 0.625, "grad_norm": 0.2658252418041229, "learning_rate": 1e-06, "loss": -0.0082, "num_tokens": 551155537.0, "reward": 0.5234375, "reward_std": 0.13041727244853973, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1354, "tools/generated_tokens": 4625.57421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.62890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1093.80078125, "completions/mean_terminated_length": 1086.287353515625, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.14498990960419178, "epoch": 0.23089867297164887, "frac_reward_zero_std": 0.625, "grad_norm": 0.25734248757362366, "learning_rate": 1e-06, "loss": 0.0253, "num_tokens": 551504926.0, "reward": 0.625, "reward_std": 0.1504075825214386, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1355, "tools/generated_tokens": 3445.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1140.63671875, "completions/mean_terminated_length": 1059.5531005859375, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.1469175275415182, "epoch": 0.2310690778963512, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2498631477355957, "learning_rate": 1e-06, "loss": -0.0031, "num_tokens": 551856625.0, "reward": 0.55078125, "reward_std": 0.09914018213748932, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1356, "tools/generated_tokens": 3604.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1155.08984375, "completions/mean_terminated_length": 1118.7926025390625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.14186188066378236, "epoch": 0.23123948282105353, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2608811855316162, "learning_rate": 1e-06, "loss": -0.0059, "num_tokens": 552221256.0, "reward": 0.62109375, "reward_std": 0.11046826094388962, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 1357, "tools/generated_tokens": 3219.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1118.33203125, "completions/mean_terminated_length": 1107.308349609375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.13944109994918108, "epoch": 0.23140988774575585, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3416125178337097, "learning_rate": 1e-06, "loss": -0.0206, "num_tokens": 552574269.0, "reward": 0.71875, "reward_std": 0.19918766617774963, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 1358, "tools/generated_tokens": 3166.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1120.97265625, "completions/mean_terminated_length": 1050.8614501953125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.160469857044518, "epoch": 0.23158029267045818, "frac_reward_zero_std": 0.5, "grad_norm": 0.28715869784355164, "learning_rate": 1e-06, "loss": 0.0166, "num_tokens": 552940342.0, "reward": 0.546875, "reward_std": 0.17396603524684906, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1359, "tools/generated_tokens": 4264.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1268.58203125, "completions/mean_terminated_length": 1145.1448974609375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.15699640568345785, "epoch": 0.2317506975951605, "frac_reward_zero_std": 0.5, "grad_norm": 0.3017316460609436, "learning_rate": 1e-06, "loss": 0.0352, "num_tokens": 553337419.0, "reward": 0.54296875, "reward_std": 0.1935332715511322, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1360, "tools/generated_tokens": 4620.5859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.63671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1196.3203125, "completions/mean_terminated_length": 1172.37744140625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.14143207715824246, "epoch": 0.23192110251986284, "frac_reward_zero_std": 0.625, "grad_norm": 0.25447598099708557, "learning_rate": 1e-06, "loss": 0.0228, "num_tokens": 553711181.0, "reward": 0.5703125, "reward_std": 0.1290597915649414, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1361, "tools/generated_tokens": 3644.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1073.93359375, "completions/mean_terminated_length": 995.8438110351562, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.14906241744756699, "epoch": 0.23209150744456514, "frac_reward_zero_std": 0.375, "grad_norm": 0.43968793749809265, "learning_rate": 1e-06, "loss": 0.0404, "num_tokens": 554071260.0, "reward": 0.578125, "reward_std": 0.23471032083034515, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1362, "tools/generated_tokens": 4177.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1103.33984375, "completions/mean_terminated_length": 1044.5435791015625, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.13886640733107924, "epoch": 0.23226191236926746, "frac_reward_zero_std": 0.625, "grad_norm": 0.23838205635547638, "learning_rate": 1e-06, "loss": -0.0242, "num_tokens": 554419155.0, "reward": 0.640625, "reward_std": 0.1580354869365692, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1363, "tools/generated_tokens": 3327.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1086.1328125, "completions/mean_terminated_length": 933.8145141601562, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.1525264997035265, "epoch": 0.2324323172939698, "frac_reward_zero_std": 0.4375, "grad_norm": 0.354211688041687, "learning_rate": 1e-06, "loss": 0.0669, "num_tokens": 554779973.0, "reward": 0.625, "reward_std": 0.23682370781898499, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1364, "tools/generated_tokens": 4382.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1287.05859375, "completions/mean_terminated_length": 1115.9425048828125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.15940591879189014, "epoch": 0.23260272221867212, "frac_reward_zero_std": 0.375, "grad_norm": 0.34133151173591614, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 555196244.0, "reward": 0.34375, "reward_std": 0.24262793362140656, "rewards/simpleverify_reward/mean": 0.34375, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1365, "tools/generated_tokens": 5199.06640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1155.09375, "completions/mean_terminated_length": 1058.4632568359375, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.1359103391878307, "epoch": 0.23277312714337445, "frac_reward_zero_std": 0.625, "grad_norm": 0.3040052056312561, "learning_rate": 1e-06, "loss": 0.0252, "num_tokens": 555568460.0, "reward": 0.57421875, "reward_std": 0.15936589241027832, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1366, "tools/generated_tokens": 4059.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1213.0703125, "completions/mean_terminated_length": 1085.1982421875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.1569506535306573, "epoch": 0.23294353206807678, "frac_reward_zero_std": 0.4375, "grad_norm": 0.40142616629600525, "learning_rate": 1e-06, "loss": 0.0272, "num_tokens": 555965214.0, "reward": 0.48046875, "reward_std": 0.24299238622188568, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1367, "tools/generated_tokens": 4613.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1278.2109375, "completions/mean_terminated_length": 1091.368896484375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.15317155700176954, "epoch": 0.2331139369927791, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3221870958805084, "learning_rate": 1e-06, "loss": 0.0387, "num_tokens": 556369860.0, "reward": 0.40625, "reward_std": 0.22553853690624237, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1368, "tools/generated_tokens": 4430.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1281.17578125, "completions/mean_terminated_length": 1122.0283203125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.15521557163447142, "epoch": 0.2332843419174814, "frac_reward_zero_std": 0.4375, "grad_norm": 0.358614057302475, "learning_rate": 1e-06, "loss": 0.0256, "num_tokens": 556772785.0, "reward": 0.62890625, "reward_std": 0.2035529911518097, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1369, "tools/generated_tokens": 4233.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1041.12890625, "completions/mean_terminated_length": 902.4044799804688, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.15183686651289463, "epoch": 0.23345474684218373, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3526458442211151, "learning_rate": 1e-06, "loss": 0.0365, "num_tokens": 557128418.0, "reward": 0.65234375, "reward_std": 0.2309720814228058, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1370, "tools/generated_tokens": 4561.125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1272.390625, "completions/mean_terminated_length": 1074.691162109375, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.15834901668131351, "epoch": 0.23362515176688606, "frac_reward_zero_std": 0.3125, "grad_norm": 0.316895991563797, "learning_rate": 1e-06, "loss": 0.0265, "num_tokens": 557545350.0, "reward": 0.5234375, "reward_std": 0.251853883266449, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1371, "tools/generated_tokens": 5456.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1161.7265625, "completions/mean_terminated_length": 1057.2314453125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.14726852346211672, "epoch": 0.23379555669158839, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3289720416069031, "learning_rate": 1e-06, "loss": 0.0373, "num_tokens": 557924496.0, "reward": 0.59765625, "reward_std": 0.23582594096660614, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1372, "tools/generated_tokens": 4417.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1200.66796875, "completions/mean_terminated_length": 1108.96533203125, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.14439208805561066, "epoch": 0.2339659616162907, "frac_reward_zero_std": 0.5, "grad_norm": 0.33840566873550415, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 558312123.0, "reward": 0.6171875, "reward_std": 0.17749404907226562, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1373, "tools/generated_tokens": 4592.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1210.9765625, "completions/mean_terminated_length": 1095.6533203125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.15194358304142952, "epoch": 0.23413636654099304, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28937697410583496, "learning_rate": 1e-06, "loss": -0.0377, "num_tokens": 558699269.0, "reward": 0.484375, "reward_std": 0.22930431365966797, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1374, "tools/generated_tokens": 4274.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1151.40234375, "completions/mean_terminated_length": 1087.6275634765625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.1511518396437168, "epoch": 0.23430677146569537, "frac_reward_zero_std": 0.625, "grad_norm": 0.24326343834400177, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 559067612.0, "reward": 0.69140625, "reward_std": 0.15474742650985718, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1375, "tools/generated_tokens": 3623.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 931.6953125, "completions/mean_terminated_length": 909.4581909179688, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.14881728868931532, "epoch": 0.2344771763903977, "frac_reward_zero_std": 0.625, "grad_norm": 0.28731590509414673, "learning_rate": 1e-06, "loss": -0.0171, "num_tokens": 559392206.0, "reward": 0.5859375, "reward_std": 0.12136821448802948, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1376, "tools/generated_tokens": 3523.69140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1059.28125, "completions/mean_terminated_length": 975.4915161132812, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.13303035032004118, "epoch": 0.2346475813151, "frac_reward_zero_std": 0.5, "grad_norm": 0.2646699845790863, "learning_rate": 1e-06, "loss": 0.0189, "num_tokens": 559744374.0, "reward": 0.625, "reward_std": 0.1642879694700241, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1377, "tools/generated_tokens": 3923.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1260.6328125, "completions/mean_terminated_length": 1097.217041015625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1482251863926649, "epoch": 0.23481798623980232, "frac_reward_zero_std": 0.375, "grad_norm": 0.44757696986198425, "learning_rate": 1e-06, "loss": 0.0352, "num_tokens": 560144600.0, "reward": 0.4609375, "reward_std": 0.23590734601020813, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1378, "tools/generated_tokens": 4516.62890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1201.81640625, "completions/mean_terminated_length": 1049.7373046875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.14814107306301594, "epoch": 0.23498839116450465, "frac_reward_zero_std": 0.4375, "grad_norm": 0.29998618364334106, "learning_rate": 1e-06, "loss": 0.0114, "num_tokens": 560534361.0, "reward": 0.515625, "reward_std": 0.18199022114276886, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1379, "tools/generated_tokens": 4609.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1187.25, "completions/mean_terminated_length": 1032.552978515625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.15758098103106022, "epoch": 0.23515879608920698, "frac_reward_zero_std": 0.625, "grad_norm": 0.3840247392654419, "learning_rate": 1e-06, "loss": 0.0261, "num_tokens": 560917817.0, "reward": 0.47265625, "reward_std": 0.16494406759738922, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1380, "tools/generated_tokens": 4859.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.79296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1238.09765625, "completions/mean_terminated_length": 1088.11572265625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.14428445417433977, "epoch": 0.2353292010139093, "frac_reward_zero_std": 0.5625, "grad_norm": 0.25626885890960693, "learning_rate": 1e-06, "loss": -0.0103, "num_tokens": 561310466.0, "reward": 0.5078125, "reward_std": 0.15746080875396729, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1381, "tools/generated_tokens": 4430.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1250.9765625, "completions/mean_terminated_length": 1133.031494140625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.1478255707770586, "epoch": 0.23549960593861163, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3447025716304779, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 561708044.0, "reward": 0.44921875, "reward_std": 0.24077895283699036, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1382, "tools/generated_tokens": 4282.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1116.39453125, "completions/mean_terminated_length": 1066.5555419921875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.1417993986979127, "epoch": 0.23567001086331396, "frac_reward_zero_std": 0.3125, "grad_norm": 0.3687669634819031, "learning_rate": 1e-06, "loss": 0.027, "num_tokens": 562075921.0, "reward": 0.66015625, "reward_std": 0.23502904176712036, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1383, "tools/generated_tokens": 4092.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1284.73828125, "completions/mean_terminated_length": 1094.8536376953125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1434822091832757, "epoch": 0.23584041578801626, "frac_reward_zero_std": 0.625, "grad_norm": 0.20903094112873077, "learning_rate": 1e-06, "loss": 0.0406, "num_tokens": 562478910.0, "reward": 0.53515625, "reward_std": 0.14359626173973083, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1384, "tools/generated_tokens": 4892.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1167.0078125, "completions/mean_terminated_length": 979.1232299804688, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.14376505743712187, "epoch": 0.2360108207127186, "frac_reward_zero_std": 0.375, "grad_norm": 0.3690033257007599, "learning_rate": 1e-06, "loss": -0.0029, "num_tokens": 562861376.0, "reward": 0.5859375, "reward_std": 0.250201940536499, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1385, "tools/generated_tokens": 4919.01953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.83203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1133.4375, "completions/mean_terminated_length": 1096.2601318359375, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.14430200215429068, "epoch": 0.23618122563742092, "frac_reward_zero_std": 0.6875, "grad_norm": 0.21483708918094635, "learning_rate": 1e-06, "loss": 0.0169, "num_tokens": 563229600.0, "reward": 0.71875, "reward_std": 0.10981409251689911, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 1386, "tools/generated_tokens": 3373.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1196.54296875, "completions/mean_terminated_length": 1112.4935302734375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.16254641953855753, "epoch": 0.23635163056212324, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2961871922016144, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 563613259.0, "reward": 0.4609375, "reward_std": 0.15888862311840057, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1387, "tools/generated_tokens": 4444.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1396.61328125, "completions/mean_terminated_length": 1192.84619140625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.13543740287423134, "epoch": 0.23652203548682557, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2567903399467468, "learning_rate": 1e-06, "loss": 0.0333, "num_tokens": 564051816.0, "reward": 0.51171875, "reward_std": 0.16904202103614807, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1388, "tools/generated_tokens": 5204.609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1060.83203125, "completions/mean_terminated_length": 1045.1627197265625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.16564824897795916, "epoch": 0.2366924404115279, "frac_reward_zero_std": 0.4375, "grad_norm": 0.32802814245224, "learning_rate": 1e-06, "loss": -0.0161, "num_tokens": 564398477.0, "reward": 0.64453125, "reward_std": 0.19979894161224365, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1389, "tools/generated_tokens": 3940.83984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1150.0859375, "completions/mean_terminated_length": 1121.1209716796875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.13747853133827448, "epoch": 0.23686284533623023, "frac_reward_zero_std": 0.4375, "grad_norm": 0.29831549525260925, "learning_rate": 1e-06, "loss": 0.04, "num_tokens": 564761139.0, "reward": 0.7109375, "reward_std": 0.2079564929008484, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1390, "tools/generated_tokens": 3654.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1062.81640625, "completions/mean_terminated_length": 1005.822265625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.13980166194960475, "epoch": 0.23703325026093255, "frac_reward_zero_std": 0.75, "grad_norm": 0.1462705284357071, "learning_rate": 1e-06, "loss": 0.0016, "num_tokens": 565110916.0, "reward": 0.578125, "reward_std": 0.07394562661647797, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1391, "tools/generated_tokens": 3782.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2010.0, "completions/mean_length": 1201.8359375, "completions/mean_terminated_length": 1130.1270751953125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.1441807495430112, "epoch": 0.23720365518563485, "frac_reward_zero_std": 0.5, "grad_norm": 0.31169331073760986, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 565497786.0, "reward": 0.73828125, "reward_std": 0.17770427465438843, "rewards/simpleverify_reward/mean": 0.73828125, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 1392, "tools/generated_tokens": 3857.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1236.11328125, "completions/mean_terminated_length": 1124.25341796875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.16952583380043507, "epoch": 0.23737406011033718, "frac_reward_zero_std": 0.4375, "grad_norm": 0.32996633648872375, "learning_rate": 1e-06, "loss": -0.0348, "num_tokens": 565895719.0, "reward": 0.37109375, "reward_std": 0.2366790771484375, "rewards/simpleverify_reward/mean": 0.37109375, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1393, "tools/generated_tokens": 5148.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1216.53515625, "completions/mean_terminated_length": 1093.5068359375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.14658395014703274, "epoch": 0.2375444650350395, "frac_reward_zero_std": 0.4375, "grad_norm": 0.31809747219085693, "learning_rate": 1e-06, "loss": 0.0227, "num_tokens": 566286384.0, "reward": 0.62890625, "reward_std": 0.21981573104858398, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1394, "tools/generated_tokens": 4664.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1005.19921875, "completions/mean_terminated_length": 958.3836059570312, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1345774275250733, "epoch": 0.23771486995974184, "frac_reward_zero_std": 0.75, "grad_norm": 0.18394610285758972, "learning_rate": 1e-06, "loss": 0.0156, "num_tokens": 566606467.0, "reward": 0.62890625, "reward_std": 0.09452171623706818, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1395, "tools/generated_tokens": 3333.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 935.45703125, "completions/mean_terminated_length": 894.9190673828125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.16237900033593178, "epoch": 0.23788527488444416, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4017430543899536, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 566932632.0, "reward": 0.796875, "reward_std": 0.24706044793128967, "rewards/simpleverify_reward/mean": 0.796875, "rewards/simpleverify_reward/std": 0.40311288833618164, "step": 1396, "tools/generated_tokens": 3903.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1151.21875, "completions/mean_terminated_length": 1013.8739013671875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.15538856200873852, "epoch": 0.2380556798091465, "frac_reward_zero_std": 0.5, "grad_norm": 0.3188590407371521, "learning_rate": 1e-06, "loss": -0.0031, "num_tokens": 567300688.0, "reward": 0.44921875, "reward_std": 0.21613198518753052, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1397, "tools/generated_tokens": 4519.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.64453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1076.04296875, "completions/mean_terminated_length": 1015.5477905273438, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.18928628414869308, "epoch": 0.23822608473384882, "frac_reward_zero_std": 0.4375, "grad_norm": 0.320904940366745, "learning_rate": 1e-06, "loss": 0.0283, "num_tokens": 567658011.0, "reward": 0.41796875, "reward_std": 0.18929657340049744, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1398, "tools/generated_tokens": 4516.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1168.625, "completions/mean_terminated_length": 1110.0, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.15542860236018896, "epoch": 0.23839648965855112, "frac_reward_zero_std": 0.625, "grad_norm": 0.32362598180770874, "learning_rate": 1e-06, "loss": -0.0088, "num_tokens": 568020763.0, "reward": 0.6328125, "reward_std": 0.1410640925168991, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1399, "tools/generated_tokens": 3488.62890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1129.3515625, "completions/mean_terminated_length": 1038.6695556640625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.14643107634037733, "epoch": 0.23856689458325345, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2918884754180908, "learning_rate": 1e-06, "loss": 0.0067, "num_tokens": 568398677.0, "reward": 0.5546875, "reward_std": 0.16207927465438843, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1400, "tools/generated_tokens": 4697.34765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1170.69921875, "completions/mean_terminated_length": 1062.9605712890625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.14248238876461983, "epoch": 0.23873729950795577, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3574035167694092, "learning_rate": 1e-06, "loss": 0.032, "num_tokens": 568775832.0, "reward": 0.55859375, "reward_std": 0.20411168038845062, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1401, "tools/generated_tokens": 4218.6875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1153.80078125, "completions/mean_terminated_length": 1094.1875, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.169779890216887, "epoch": 0.2389077044326581, "frac_reward_zero_std": 0.5, "grad_norm": 0.31990957260131836, "learning_rate": 1e-06, "loss": 0.0341, "num_tokens": 569156853.0, "reward": 0.7109375, "reward_std": 0.19508779048919678, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1402, "tools/generated_tokens": 4393.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1255.51953125, "completions/mean_terminated_length": 1134.148681640625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.14038816234096885, "epoch": 0.23907810935736043, "frac_reward_zero_std": 0.625, "grad_norm": 0.3096086382865906, "learning_rate": 1e-06, "loss": 0.0099, "num_tokens": 569558842.0, "reward": 0.3984375, "reward_std": 0.13896197080612183, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1403, "tools/generated_tokens": 4647.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1140.81640625, "completions/mean_terminated_length": 1122.7449951171875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.154086796566844, "epoch": 0.23924851428206276, "frac_reward_zero_std": 0.8125, "grad_norm": 0.1539401113986969, "learning_rate": 1e-06, "loss": -0.0022, "num_tokens": 569927419.0, "reward": 0.578125, "reward_std": 0.07309441268444061, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1404, "tools/generated_tokens": 3820.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1102.50390625, "completions/mean_terminated_length": 1072.0040283203125, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.15138112381100655, "epoch": 0.23941891920676509, "frac_reward_zero_std": 0.5625, "grad_norm": 0.28858381509780884, "learning_rate": 1e-06, "loss": 0.0029, "num_tokens": 570295532.0, "reward": 0.54296875, "reward_std": 0.15976692736148834, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1405, "tools/generated_tokens": 4086.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1119.9609375, "completions/mean_terminated_length": 972.9864501953125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.1423200168646872, "epoch": 0.2395893241314674, "frac_reward_zero_std": 0.375, "grad_norm": 0.3798171281814575, "learning_rate": 1e-06, "loss": 0.0342, "num_tokens": 570670210.0, "reward": 0.53515625, "reward_std": 0.24722152948379517, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1406, "tools/generated_tokens": 4719.9609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1093.89453125, "completions/mean_terminated_length": 1030.28759765625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1326984199695289, "epoch": 0.2397597290561697, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3431290090084076, "learning_rate": 1e-06, "loss": 0.0164, "num_tokens": 571020039.0, "reward": 0.4296875, "reward_std": 0.14242157340049744, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1407, "tools/generated_tokens": 3629.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1075.51953125, "completions/mean_terminated_length": 1001.9706420898438, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.15770272351801395, "epoch": 0.23993013398087204, "frac_reward_zero_std": 0.5, "grad_norm": 0.3825744092464447, "learning_rate": 1e-06, "loss": 0.0324, "num_tokens": 571374300.0, "reward": 0.39453125, "reward_std": 0.2142539918422699, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1408, "tools/generated_tokens": 4203.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1160.9140625, "completions/mean_terminated_length": 1064.9090576171875, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.1503489352762699, "epoch": 0.24010053890557437, "frac_reward_zero_std": 0.5, "grad_norm": 0.323911190032959, "learning_rate": 1e-06, "loss": 0.0451, "num_tokens": 571752582.0, "reward": 0.5, "reward_std": 0.21171057224273682, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1409, "tools/generated_tokens": 4336.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1114.0234375, "completions/mean_terminated_length": 1051.7584228515625, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.1511819064617157, "epoch": 0.2402709438302767, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3393208086490631, "learning_rate": 1e-06, "loss": 0.0423, "num_tokens": 572117964.0, "reward": 0.49609375, "reward_std": 0.21458663046360016, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1410, "tools/generated_tokens": 4242.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.52734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1177.671875, "completions/mean_terminated_length": 1115.765625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.15551136434078217, "epoch": 0.24044134875497902, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4248116910457611, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 572497320.0, "reward": 0.6796875, "reward_std": 0.15569132566452026, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1411, "tools/generated_tokens": 4393.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1157.25390625, "completions/mean_terminated_length": 1006.7625122070312, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1464686021208763, "epoch": 0.24061175367968135, "frac_reward_zero_std": 0.3125, "grad_norm": 0.41909265518188477, "learning_rate": 1e-06, "loss": -0.0027, "num_tokens": 572875801.0, "reward": 0.45703125, "reward_std": 0.26808953285217285, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1412, "tools/generated_tokens": 5005.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1136.83984375, "completions/mean_terminated_length": 1095.9305419921875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.15223989449441433, "epoch": 0.24078215860438368, "frac_reward_zero_std": 0.5, "grad_norm": 0.29868748784065247, "learning_rate": 1e-06, "loss": 0.0356, "num_tokens": 573238240.0, "reward": 0.5703125, "reward_std": 0.1937704086303711, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1413, "tools/generated_tokens": 3856.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 992.46484375, "completions/mean_terminated_length": 967.1320190429688, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.14023830648511648, "epoch": 0.24095256352908598, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3060653507709503, "learning_rate": 1e-06, "loss": 0.0189, "num_tokens": 573570391.0, "reward": 0.47265625, "reward_std": 0.22830653190612793, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1414, "tools/generated_tokens": 3592.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1126.27734375, "completions/mean_terminated_length": 1008.524169921875, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.14358393428847194, "epoch": 0.2411229684537883, "frac_reward_zero_std": 0.75, "grad_norm": 0.20425577461719513, "learning_rate": 1e-06, "loss": -0.0127, "num_tokens": 573935006.0, "reward": 0.40234375, "reward_std": 0.08912044763565063, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1415, "tools/generated_tokens": 4150.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1003.47265625, "completions/mean_terminated_length": 974.1083984375, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.16764251049607992, "epoch": 0.24129337337849063, "frac_reward_zero_std": 0.25, "grad_norm": 0.3968813419342041, "learning_rate": 1e-06, "loss": -0.0337, "num_tokens": 574298487.0, "reward": 0.5625, "reward_std": 0.2603302001953125, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1416, "tools/generated_tokens": 3787.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1236.44921875, "completions/mean_terminated_length": 1171.38818359375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.15007042838260531, "epoch": 0.24146377830319296, "frac_reward_zero_std": 0.5, "grad_norm": 0.34556326270103455, "learning_rate": 1e-06, "loss": 0.002, "num_tokens": 574692826.0, "reward": 0.6328125, "reward_std": 0.20818254351615906, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1417, "tools/generated_tokens": 4116.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1237.51953125, "completions/mean_terminated_length": 1172.544189453125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.1450265310704708, "epoch": 0.2416341832278953, "frac_reward_zero_std": 0.4375, "grad_norm": 0.35437247157096863, "learning_rate": 1e-06, "loss": 0.0105, "num_tokens": 575082831.0, "reward": 0.64453125, "reward_std": 0.20291273295879364, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1418, "tools/generated_tokens": 4309.515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1059.84375, "completions/mean_terminated_length": 1032.064208984375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.1529227653518319, "epoch": 0.24180458815259762, "frac_reward_zero_std": 0.6875, "grad_norm": 0.24530482292175293, "learning_rate": 1e-06, "loss": 0.0041, "num_tokens": 575432471.0, "reward": 0.640625, "reward_std": 0.11179865896701813, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1419, "tools/generated_tokens": 3275.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.08203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1204.03515625, "completions/mean_terminated_length": 1144.004150390625, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.1466690246015787, "epoch": 0.24197499307729994, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3316318690776825, "learning_rate": 1e-06, "loss": 0.0356, "num_tokens": 575804192.0, "reward": 0.6796875, "reward_std": 0.11664125323295593, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1420, "tools/generated_tokens": 3556.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1211.5546875, "completions/mean_terminated_length": 1083.4549560546875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.14036338403820992, "epoch": 0.24214539800200227, "frac_reward_zero_std": 0.5, "grad_norm": 0.3116992115974426, "learning_rate": 1e-06, "loss": 0.0077, "num_tokens": 576190894.0, "reward": 0.65234375, "reward_std": 0.19414390623569489, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1421, "tools/generated_tokens": 4115.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1111.41796875, "completions/mean_terminated_length": 1061.312744140625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.13980505242943764, "epoch": 0.24231580292670457, "frac_reward_zero_std": 0.375, "grad_norm": 0.3506612181663513, "learning_rate": 1e-06, "loss": 0.0075, "num_tokens": 576547673.0, "reward": 0.640625, "reward_std": 0.2275887131690979, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1422, "tools/generated_tokens": 4247.41796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1332.57421875, "completions/mean_terminated_length": 1237.606201171875, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.14347519166767597, "epoch": 0.2424862078514069, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28963902592658997, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 576972972.0, "reward": 0.6171875, "reward_std": 0.21511822938919067, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1423, "tools/generated_tokens": 4612.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1128.8125, "completions/mean_terminated_length": 1055.122314453125, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.15965467412024736, "epoch": 0.24265661277610923, "frac_reward_zero_std": 0.5625, "grad_norm": 0.28109264373779297, "learning_rate": 1e-06, "loss": 0.0286, "num_tokens": 577343724.0, "reward": 0.7109375, "reward_std": 0.18574902415275574, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1424, "tools/generated_tokens": 4184.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1056.8828125, "completions/mean_terminated_length": 1003.8600463867188, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.13751173950731754, "epoch": 0.24282701770081155, "frac_reward_zero_std": 0.5, "grad_norm": 0.35235586762428284, "learning_rate": 1e-06, "loss": 0.0136, "num_tokens": 577681710.0, "reward": 0.59375, "reward_std": 0.21393243968486786, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1425, "tools/generated_tokens": 3760.875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1031.02734375, "completions/mean_terminated_length": 1010.7689819335938, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.15521914139389992, "epoch": 0.24299742262551388, "frac_reward_zero_std": 0.375, "grad_norm": 0.4569633901119232, "learning_rate": 1e-06, "loss": -0.0051, "num_tokens": 578009989.0, "reward": 0.52734375, "reward_std": 0.22523343563079834, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1426, "tools/generated_tokens": 3351.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1305.203125, "completions/mean_terminated_length": 1187.565673828125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.15778795164078474, "epoch": 0.2431678275502162, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3568911850452423, "learning_rate": 1e-06, "loss": 0.0258, "num_tokens": 578421673.0, "reward": 0.56640625, "reward_std": 0.14547231793403625, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1427, "tools/generated_tokens": 4345.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1182.23046875, "completions/mean_terminated_length": 1120.6485595703125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.14691222785040736, "epoch": 0.24333823247491854, "frac_reward_zero_std": 0.4375, "grad_norm": 0.28507331013679504, "learning_rate": 1e-06, "loss": -0.0095, "num_tokens": 578796324.0, "reward": 0.5625, "reward_std": 0.20346032083034515, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1428, "tools/generated_tokens": 3766.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1236.63671875, "completions/mean_terminated_length": 1144.921630859375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.15272565651685, "epoch": 0.24350863739962084, "frac_reward_zero_std": 0.3125, "grad_norm": 0.37567347288131714, "learning_rate": 1e-06, "loss": -0.0119, "num_tokens": 579179127.0, "reward": 0.7109375, "reward_std": 0.2568970322608948, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1429, "tools/generated_tokens": 3940.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1086.97265625, "completions/mean_terminated_length": 1035.5596923828125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.1435585650615394, "epoch": 0.24367904232432316, "frac_reward_zero_std": 0.375, "grad_norm": 0.3296862244606018, "learning_rate": 1e-06, "loss": -0.0012, "num_tokens": 579528256.0, "reward": 0.4921875, "reward_std": 0.25789350271224976, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1430, "tools/generated_tokens": 3942.9765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1179.9375, "completions/mean_terminated_length": 1129.72314453125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.13656746316701174, "epoch": 0.2438494472490255, "frac_reward_zero_std": 0.375, "grad_norm": 0.49334725737571716, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 579906224.0, "reward": 0.4921875, "reward_std": 0.23557472229003906, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1431, "tools/generated_tokens": 4187.9453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1203.6484375, "completions/mean_terminated_length": 1116.3060302734375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.13039688859134912, "epoch": 0.24401985217372782, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3494527339935303, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 580294934.0, "reward": 0.640625, "reward_std": 0.1982850879430771, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1432, "tools/generated_tokens": 4419.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1006.7734375, "completions/mean_terminated_length": 981.7840576171875, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.13485337980091572, "epoch": 0.24419025709843015, "frac_reward_zero_std": 0.625, "grad_norm": 0.2833566963672638, "learning_rate": 1e-06, "loss": 0.0009, "num_tokens": 580625052.0, "reward": 0.51171875, "reward_std": 0.15224426984786987, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1433, "tools/generated_tokens": 3406.77734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1054.98046875, "completions/mean_terminated_length": 997.5371704101562, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.13075955538079143, "epoch": 0.24436066202313247, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4090021252632141, "learning_rate": 1e-06, "loss": -0.0263, "num_tokens": 580978231.0, "reward": 0.59765625, "reward_std": 0.26856935024261475, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1434, "tools/generated_tokens": 4126.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1118.9453125, "completions/mean_terminated_length": 1104.198486328125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1303746779449284, "epoch": 0.2445310669478348, "frac_reward_zero_std": 0.6875, "grad_norm": 0.22322238981723785, "learning_rate": 1e-06, "loss": -0.018, "num_tokens": 581336169.0, "reward": 0.44140625, "reward_std": 0.11838587373495102, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1435, "tools/generated_tokens": 3782.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1037.97265625, "completions/mean_terminated_length": 1030.0196533203125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.1261001848615706, "epoch": 0.24470147187253713, "frac_reward_zero_std": 0.8125, "grad_norm": 0.1705537885427475, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 581658674.0, "reward": 0.7890625, "reward_std": 0.08157352358102798, "rewards/simpleverify_reward/mean": 0.7890625, "rewards/simpleverify_reward/std": 0.4087733030319214, "step": 1436, "tools/generated_tokens": 2773.96875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.84765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 963.28515625, "completions/mean_terminated_length": 963.28515625, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.12777064880356193, "epoch": 0.24487187679723943, "frac_reward_zero_std": 0.625, "grad_norm": 0.31700143218040466, "learning_rate": 1e-06, "loss": 0.0149, "num_tokens": 581984523.0, "reward": 0.66015625, "reward_std": 0.1467868983745575, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1437, "tools/generated_tokens": 3323.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1091.19140625, "completions/mean_terminated_length": 1068.22802734375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.11948351096361876, "epoch": 0.24504228172194176, "frac_reward_zero_std": 0.375, "grad_norm": 0.3786174952983856, "learning_rate": 1e-06, "loss": -0.0231, "num_tokens": 582349676.0, "reward": 0.5625, "reward_std": 0.2506016492843628, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1438, "tools/generated_tokens": 4003.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1062.7890625, "completions/mean_terminated_length": 1001.4730834960938, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.12576998071745038, "epoch": 0.24521268664664408, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2762691080570221, "learning_rate": 1e-06, "loss": 0.0527, "num_tokens": 582699014.0, "reward": 0.54296875, "reward_std": 0.16505970060825348, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1439, "tools/generated_tokens": 4390.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1094.65625, "completions/mean_terminated_length": 1071.7760009765625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.12418131623417139, "epoch": 0.2453830915713464, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4065541923046112, "learning_rate": 1e-06, "loss": -0.0046, "num_tokens": 583062046.0, "reward": 0.5546875, "reward_std": 0.2620996832847595, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1440, "tools/generated_tokens": 3350.64453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1211.78125, "completions/mean_terminated_length": 1133.1624755859375, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.12346707889810205, "epoch": 0.24555349649604874, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2874394953250885, "learning_rate": 1e-06, "loss": 0.019, "num_tokens": 583452694.0, "reward": 0.546875, "reward_std": 0.16542133688926697, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1441, "tools/generated_tokens": 4075.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1078.66015625, "completions/mean_terminated_length": 1005.3488159179688, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12281754519790411, "epoch": 0.24572390142075107, "frac_reward_zero_std": 0.75, "grad_norm": 0.23336999118328094, "learning_rate": 1e-06, "loss": -0.0029, "num_tokens": 583796895.0, "reward": 0.55859375, "reward_std": 0.10638156533241272, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1442, "tools/generated_tokens": 3398.6171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1184.63671875, "completions/mean_terminated_length": 1177.838623046875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.11237064562737942, "epoch": 0.2458943063454534, "frac_reward_zero_std": 0.375, "grad_norm": 0.3196418285369873, "learning_rate": 1e-06, "loss": -0.0045, "num_tokens": 584164850.0, "reward": 0.6171875, "reward_std": 0.22798693180084229, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1443, "tools/generated_tokens": 2640.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2020.0, "completions/mean_length": 1015.03125, "completions/mean_terminated_length": 955.2809448242188, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.11015663156285882, "epoch": 0.2460647112701557, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3735048472881317, "learning_rate": 1e-06, "loss": 0.0045, "num_tokens": 584491178.0, "reward": 0.375, "reward_std": 0.17703913152217865, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1444, "tools/generated_tokens": 3879.0390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1075.0390625, "completions/mean_terminated_length": 1047.686767578125, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.1327337739057839, "epoch": 0.24623511619485802, "frac_reward_zero_std": 0.5, "grad_norm": 0.6685805320739746, "learning_rate": 1e-06, "loss": 0.0078, "num_tokens": 584839876.0, "reward": 0.578125, "reward_std": 0.1624118983745575, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1445, "tools/generated_tokens": 3379.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1087.8125, "completions/mean_terminated_length": 1044.7060546875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.10810331907123327, "epoch": 0.24640552111956035, "frac_reward_zero_std": 0.75, "grad_norm": 0.23567412793636322, "learning_rate": 1e-06, "loss": 0.0109, "num_tokens": 585191572.0, "reward": 0.40234375, "reward_std": 0.09287451207637787, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1446, "tools/generated_tokens": 3607.82421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 934.95703125, "completions/mean_terminated_length": 930.5922241210938, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.12249541282653809, "epoch": 0.24657592604426268, "frac_reward_zero_std": 0.5, "grad_norm": 0.48707854747772217, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 585507657.0, "reward": 0.58984375, "reward_std": 0.1902293711900711, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1447, "tools/generated_tokens": 3182.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1104.68359375, "completions/mean_terminated_length": 1078.1646728515625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.10838250489905477, "epoch": 0.246746330968965, "frac_reward_zero_std": 0.5, "grad_norm": 0.2959323823451996, "learning_rate": 1e-06, "loss": -0.0062, "num_tokens": 585859928.0, "reward": 0.62890625, "reward_std": 0.1750703752040863, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1448, "tools/generated_tokens": 3720.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1155.3828125, "completions/mean_terminated_length": 1115.3060302734375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11246175831183791, "epoch": 0.24691673589366733, "frac_reward_zero_std": 0.375, "grad_norm": 0.4171391725540161, "learning_rate": 1e-06, "loss": 0.0103, "num_tokens": 586220298.0, "reward": 0.5546875, "reward_std": 0.24625971913337708, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1449, "tools/generated_tokens": 3691.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 1128.8125, "completions/mean_terminated_length": 1055.122314453125, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.11687920754775405, "epoch": 0.24708714081836966, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5016071796417236, "learning_rate": 1e-06, "loss": 0.0237, "num_tokens": 586579482.0, "reward": 0.5078125, "reward_std": 0.1569422334432602, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1450, "tools/generated_tokens": 3984.81640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1088.4296875, "completions/mean_terminated_length": 1061.4537353515625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.11327971518039703, "epoch": 0.247257545743072, "frac_reward_zero_std": 0.5, "grad_norm": 0.41301220655441284, "learning_rate": 1e-06, "loss": -0.0016, "num_tokens": 586931512.0, "reward": 0.45703125, "reward_std": 0.18543875217437744, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1451, "tools/generated_tokens": 3640.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1036.53515625, "completions/mean_terminated_length": 1012.2600708007812, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.11719759367406368, "epoch": 0.2474279506677743, "frac_reward_zero_std": 0.5625, "grad_norm": 0.35212767124176025, "learning_rate": 1e-06, "loss": 0.0136, "num_tokens": 587268801.0, "reward": 0.703125, "reward_std": 0.18815405666828156, "rewards/simpleverify_reward/mean": 0.703125, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 1452, "tools/generated_tokens": 3340.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1083.8046875, "completions/mean_terminated_length": 1006.5062866210938, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.11095784977078438, "epoch": 0.24759835559247662, "frac_reward_zero_std": 0.4375, "grad_norm": 0.41422900557518005, "learning_rate": 1e-06, "loss": 0.0295, "num_tokens": 587616479.0, "reward": 0.51953125, "reward_std": 0.24328210949897766, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1453, "tools/generated_tokens": 4107.80859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1085.66015625, "completions/mean_terminated_length": 1008.510498046875, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.1148814195767045, "epoch": 0.24776876051717894, "frac_reward_zero_std": 0.5, "grad_norm": 0.3686800003051758, "learning_rate": 1e-06, "loss": 0.0137, "num_tokens": 587979752.0, "reward": 0.359375, "reward_std": 0.20601484179496765, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1454, "tools/generated_tokens": 4629.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1143.08984375, "completions/mean_terminated_length": 1110.117431640625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12750268820673227, "epoch": 0.24793916544188127, "frac_reward_zero_std": 0.5625, "grad_norm": 0.36884605884552, "learning_rate": 1e-06, "loss": -0.0068, "num_tokens": 588341775.0, "reward": 0.5859375, "reward_std": 0.16691280901432037, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1455, "tools/generated_tokens": 3679.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1152.06640625, "completions/mean_terminated_length": 1084.3067626953125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.10520532168447971, "epoch": 0.2481095703665836, "frac_reward_zero_std": 0.3125, "grad_norm": 0.39846134185791016, "learning_rate": 1e-06, "loss": -0.031, "num_tokens": 588714832.0, "reward": 0.6328125, "reward_std": 0.29392433166503906, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1456, "tools/generated_tokens": 4480.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1251.64453125, "completions/mean_terminated_length": 1112.8302001953125, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.12598368618637323, "epoch": 0.24827997529128593, "frac_reward_zero_std": 0.625, "grad_norm": 0.3435189425945282, "learning_rate": 1e-06, "loss": 0.0289, "num_tokens": 589116421.0, "reward": 0.41015625, "reward_std": 0.14792026579380035, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1457, "tools/generated_tokens": 5219.64453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.9375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1058.8359375, "completions/mean_terminated_length": 1031.028076171875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.11896358709782362, "epoch": 0.24845038021598825, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3003554344177246, "learning_rate": 1e-06, "loss": 0.0123, "num_tokens": 589469867.0, "reward": 0.48046875, "reward_std": 0.1356898993253708, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1458, "tools/generated_tokens": 4122.83984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1216.6015625, "completions/mean_terminated_length": 1138.440185546875, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.12667619716376066, "epoch": 0.24862078514069055, "frac_reward_zero_std": 0.375, "grad_norm": 0.42140600085258484, "learning_rate": 1e-06, "loss": -0.0053, "num_tokens": 589861013.0, "reward": 0.4296875, "reward_std": 0.26385819911956787, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1459, "tools/generated_tokens": 4672.609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1129.96875, "completions/mean_terminated_length": 1126.36865234375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.14301754999905825, "epoch": 0.24879119006539288, "frac_reward_zero_std": 0.5, "grad_norm": 0.33536893129348755, "learning_rate": 1e-06, "loss": 0.0239, "num_tokens": 590223421.0, "reward": 0.65625, "reward_std": 0.1933721899986267, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1460, "tools/generated_tokens": 3481.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1153.34375, "completions/mean_terminated_length": 1128.1927490234375, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.11208975315093994, "epoch": 0.2489615949900952, "frac_reward_zero_std": 0.5, "grad_norm": 0.3024435341358185, "learning_rate": 1e-06, "loss": 0.0201, "num_tokens": 590585733.0, "reward": 0.57421875, "reward_std": 0.20270179212093353, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1461, "tools/generated_tokens": 3641.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1081.2890625, "completions/mean_terminated_length": 1050.1048583984375, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.1259052320383489, "epoch": 0.24913199991479754, "frac_reward_zero_std": 0.5, "grad_norm": 0.42138561606407166, "learning_rate": 1e-06, "loss": -0.0152, "num_tokens": 590930815.0, "reward": 0.484375, "reward_std": 0.2035800814628601, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1462, "tools/generated_tokens": 3817.29296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1017.2109375, "completions/mean_terminated_length": 1000.8492431640625, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.1401116792112589, "epoch": 0.24930240483949986, "frac_reward_zero_std": 0.375, "grad_norm": 0.4687753915786743, "learning_rate": 1e-06, "loss": 0.009, "num_tokens": 591264245.0, "reward": 0.6015625, "reward_std": 0.23756122589111328, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1463, "tools/generated_tokens": 3401.21484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 944.359375, "completions/mean_terminated_length": 926.84130859375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.12122867489233613, "epoch": 0.2494728097642022, "frac_reward_zero_std": 0.375, "grad_norm": 0.44950205087661743, "learning_rate": 1e-06, "loss": -0.0082, "num_tokens": 591581393.0, "reward": 0.578125, "reward_std": 0.25605812668800354, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1464, "tools/generated_tokens": 3624.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1181.0390625, "completions/mean_terminated_length": 1127.078857421875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.12016493873670697, "epoch": 0.24964321468890452, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3320082128047943, "learning_rate": 1e-06, "loss": -0.0061, "num_tokens": 591952267.0, "reward": 0.65234375, "reward_std": 0.15931200981140137, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1465, "tools/generated_tokens": 3717.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2045.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1091.52734375, "completions/mean_terminated_length": 1091.52734375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.13484007632359862, "epoch": 0.24981361961360685, "frac_reward_zero_std": 0.375, "grad_norm": 0.3817463517189026, "learning_rate": 1e-06, "loss": -0.0274, "num_tokens": 592312050.0, "reward": 0.71875, "reward_std": 0.2410845011472702, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 1466, "tools/generated_tokens": 3691.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2032.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1111.625, "completions/mean_terminated_length": 1111.625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.12843712512403727, "epoch": 0.24998402453830915, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4103078246116638, "learning_rate": 1e-06, "loss": 0.0286, "num_tokens": 592649090.0, "reward": 0.6875, "reward_std": 0.1816575825214386, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1467, "tools/generated_tokens": 2479.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1174.6484375, "completions/mean_terminated_length": 1104.6328125, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.11294136475771666, "epoch": 0.2501544294630115, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3989618718624115, "learning_rate": 1e-06, "loss": -0.0051, "num_tokens": 593033384.0, "reward": 0.51953125, "reward_std": 0.1892564743757248, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1468, "tools/generated_tokens": 4198.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1179.203125, "completions/mean_terminated_length": 1097.5213623046875, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.12915315059944987, "epoch": 0.2503248343877138, "frac_reward_zero_std": 0.625, "grad_norm": 0.33860647678375244, "learning_rate": 1e-06, "loss": -0.0163, "num_tokens": 593412284.0, "reward": 0.359375, "reward_std": 0.14789125323295593, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1469, "tools/generated_tokens": 4531.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.63671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1110.53515625, "completions/mean_terminated_length": 1088.0360107421875, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.13009409932419658, "epoch": 0.25049523931241613, "frac_reward_zero_std": 0.625, "grad_norm": 0.2981676161289215, "learning_rate": 1e-06, "loss": -0.0052, "num_tokens": 593778789.0, "reward": 0.54296875, "reward_std": 0.11091843992471695, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1470, "tools/generated_tokens": 3542.53515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1075.08984375, "completions/mean_terminated_length": 1035.544677734375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.12851005187258124, "epoch": 0.25066564423711846, "frac_reward_zero_std": 0.625, "grad_norm": 0.32632043957710266, "learning_rate": 1e-06, "loss": -0.0017, "num_tokens": 594113500.0, "reward": 0.45703125, "reward_std": 0.138350710272789, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1471, "tools/generated_tokens": 3403.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1074.94921875, "completions/mean_terminated_length": 1055.5657958984375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.12474041199311614, "epoch": 0.2508360491618208, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3876008093357086, "learning_rate": 1e-06, "loss": -0.0334, "num_tokens": 594468207.0, "reward": 0.53515625, "reward_std": 0.21658216416835785, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1472, "tools/generated_tokens": 3570.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 957.57421875, "completions/mean_terminated_length": 940.2659301757812, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.14804183319211006, "epoch": 0.2510064540865231, "frac_reward_zero_std": 0.4375, "grad_norm": 0.41524437069892883, "learning_rate": 1e-06, "loss": -0.0036, "num_tokens": 594787458.0, "reward": 0.66015625, "reward_std": 0.2152675986289978, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1473, "tools/generated_tokens": 3269.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1107.5234375, "completions/mean_terminated_length": 1103.8353271484375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.14154515136033297, "epoch": 0.25117685901122544, "frac_reward_zero_std": 0.5, "grad_norm": 0.29425472021102905, "learning_rate": 1e-06, "loss": 0.0035, "num_tokens": 595149368.0, "reward": 0.44921875, "reward_std": 0.1822051852941513, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1474, "tools/generated_tokens": 3835.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1037.5234375, "completions/mean_terminated_length": 1009.116455078125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.16778101585805416, "epoch": 0.25134726393592777, "frac_reward_zero_std": 0.3125, "grad_norm": 0.7174923419952393, "learning_rate": 1e-06, "loss": 0.0013, "num_tokens": 595492014.0, "reward": 0.4296875, "reward_std": 0.2783769369125366, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1475, "tools/generated_tokens": 3901.52734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 831.81640625, "completions/mean_terminated_length": 817.395263671875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.15140918362885714, "epoch": 0.2515176688606301, "frac_reward_zero_std": 0.6875, "grad_norm": 0.43679115176200867, "learning_rate": 1e-06, "loss": -0.0046, "num_tokens": 595772703.0, "reward": 0.56640625, "reward_std": 0.13204212486743927, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1476, "tools/generated_tokens": 3111.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1137.23046875, "completions/mean_terminated_length": 1122.77392578125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.15122841484844685, "epoch": 0.2516880737853324, "frac_reward_zero_std": 0.4375, "grad_norm": 0.43355467915534973, "learning_rate": 1e-06, "loss": -0.0035, "num_tokens": 596142426.0, "reward": 0.62109375, "reward_std": 0.21180322766304016, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 1477, "tools/generated_tokens": 3601.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1006.96484375, "completions/mean_terminated_length": 1002.8823852539062, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.17671490088105202, "epoch": 0.2518584787100347, "frac_reward_zero_std": 0.6875, "grad_norm": 0.27541232109069824, "learning_rate": 1e-06, "loss": -0.0027, "num_tokens": 596459777.0, "reward": 0.5078125, "reward_std": 0.137538880109787, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1478, "tools/generated_tokens": 2758.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1159.98046875, "completions/mean_terminated_length": 1100.7791748046875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.13036366505548358, "epoch": 0.252028883634737, "frac_reward_zero_std": 0.3125, "grad_norm": 0.48330244421958923, "learning_rate": 1e-06, "loss": -0.0036, "num_tokens": 596832076.0, "reward": 0.609375, "reward_std": 0.2846473157405853, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1479, "tools/generated_tokens": 4167.97265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 995.99609375, "completions/mean_terminated_length": 948.7632446289062, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.13148819003254175, "epoch": 0.25219928855943935, "frac_reward_zero_std": 0.625, "grad_norm": 0.28951647877693176, "learning_rate": 1e-06, "loss": 0.0204, "num_tokens": 597157803.0, "reward": 0.625, "reward_std": 0.12444131821393967, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1480, "tools/generated_tokens": 3379.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1149.75390625, "completions/mean_terminated_length": 1085.8619384765625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.12756054336205125, "epoch": 0.2523696934841417, "frac_reward_zero_std": 0.625, "grad_norm": 0.23577015101909637, "learning_rate": 1e-06, "loss": -0.0194, "num_tokens": 597505724.0, "reward": 0.66796875, "reward_std": 0.09947281330823898, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1481, "tools/generated_tokens": 2765.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1186.04296875, "completions/mean_terminated_length": 1116.94091796875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.12666014349088073, "epoch": 0.252540098408844, "frac_reward_zero_std": 0.5, "grad_norm": 0.3999378979206085, "learning_rate": 1e-06, "loss": 0.003, "num_tokens": 597882839.0, "reward": 0.55859375, "reward_std": 0.19881373643875122, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1482, "tools/generated_tokens": 4250.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1157.4140625, "completions/mean_terminated_length": 1121.2113037109375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.13788915565237403, "epoch": 0.25271050333354633, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3218672275543213, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 598251841.0, "reward": 0.62890625, "reward_std": 0.1624389886856079, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1483, "tools/generated_tokens": 3157.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1144.8515625, "completions/mean_terminated_length": 1115.7176513671875, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.15594645775854588, "epoch": 0.25288090825824866, "frac_reward_zero_std": 0.5625, "grad_norm": 0.9339337348937988, "learning_rate": 1e-06, "loss": 0.0311, "num_tokens": 598610971.0, "reward": 0.61328125, "reward_std": 0.16516819596290588, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1484, "tools/generated_tokens": 3232.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1160.13671875, "completions/mean_terminated_length": 1068.2930908203125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.15929285902529955, "epoch": 0.253051313182951, "frac_reward_zero_std": 0.3125, "grad_norm": 0.8043149709701538, "learning_rate": 1e-06, "loss": 0.0269, "num_tokens": 598989710.0, "reward": 0.41796875, "reward_std": 0.2640684247016907, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1485, "tools/generated_tokens": 3880.140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1000.26953125, "completions/mean_terminated_length": 962.0931396484375, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.13257735036313534, "epoch": 0.2532217181076533, "frac_reward_zero_std": 0.5, "grad_norm": 0.5502130389213562, "learning_rate": 1e-06, "loss": 0.0154, "num_tokens": 599322531.0, "reward": 0.7109375, "reward_std": 0.21951064467430115, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1486, "tools/generated_tokens": 4000.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1114.859375, "completions/mean_terminated_length": 1004.8384399414062, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.15834834147244692, "epoch": 0.25339212303235564, "frac_reward_zero_std": 0.5625, "grad_norm": 0.35030290484428406, "learning_rate": 1e-06, "loss": 0.0285, "num_tokens": 599690015.0, "reward": 0.69140625, "reward_std": 0.16368991136550903, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1487, "tools/generated_tokens": 4130.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.47265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1110.609375, "completions/mean_terminated_length": 1060.4608154296875, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.1310226498171687, "epoch": 0.25356252795705797, "frac_reward_zero_std": 0.5, "grad_norm": 0.6122882962226868, "learning_rate": 1e-06, "loss": 0.0206, "num_tokens": 600051275.0, "reward": 0.7109375, "reward_std": 0.18023644387722015, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1488, "tools/generated_tokens": 3830.6015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1046.08203125, "completions/mean_terminated_length": 1030.1785888671875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.13269883254542947, "epoch": 0.2537329328817603, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3630038797855377, "learning_rate": 1e-06, "loss": -0.0244, "num_tokens": 600390720.0, "reward": 0.4765625, "reward_std": 0.1978301852941513, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1489, "tools/generated_tokens": 3174.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1105.72265625, "completions/mean_terminated_length": 994.6244506835938, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.1352847800590098, "epoch": 0.2539033378064626, "frac_reward_zero_std": 0.625, "grad_norm": 0.3241891860961914, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 600752137.0, "reward": 0.453125, "reward_std": 0.15063363313674927, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1490, "tools/generated_tokens": 4249.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1166.6328125, "completions/mean_terminated_length": 1087.872314453125, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.13226392772048712, "epoch": 0.25407374273116495, "frac_reward_zero_std": 0.625, "grad_norm": 0.3054288923740387, "learning_rate": 1e-06, "loss": 0.0017, "num_tokens": 601124283.0, "reward": 0.57421875, "reward_std": 0.1318160742521286, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1491, "tools/generated_tokens": 3854.6328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1091.75, "completions/mean_terminated_length": 1019.4286499023438, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.12571618426591158, "epoch": 0.2542441476558673, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4321838617324829, "learning_rate": 1e-06, "loss": 0.0276, "num_tokens": 601464795.0, "reward": 0.484375, "reward_std": 0.21069888770580292, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1492, "tools/generated_tokens": 3395.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1129.79296875, "completions/mean_terminated_length": 1092.4674072265625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.14214461855590343, "epoch": 0.25441455258056955, "frac_reward_zero_std": 0.625, "grad_norm": 0.43382376432418823, "learning_rate": 1e-06, "loss": -0.0052, "num_tokens": 601835174.0, "reward": 0.53515625, "reward_std": 0.13391819596290588, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1493, "tools/generated_tokens": 4001.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1153.87109375, "completions/mean_terminated_length": 1102.14453125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.12694982159882784, "epoch": 0.2545849575052719, "frac_reward_zero_std": 0.5625, "grad_norm": 0.25308889150619507, "learning_rate": 1e-06, "loss": -0.0372, "num_tokens": 602198341.0, "reward": 0.67578125, "reward_std": 0.1428087055683136, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1494, "tools/generated_tokens": 3417.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1163.05078125, "completions/mean_terminated_length": 1100.1046142578125, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.12254002317786217, "epoch": 0.2547553624299742, "frac_reward_zero_std": 0.4375, "grad_norm": 0.33974984288215637, "learning_rate": 1e-06, "loss": -0.0174, "num_tokens": 602573298.0, "reward": 0.59765625, "reward_std": 0.21578145027160645, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1495, "tools/generated_tokens": 3435.05078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1172.80078125, "completions/mean_terminated_length": 1155.3665771484375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.13271049177274108, "epoch": 0.25492576735467654, "frac_reward_zero_std": 0.4375, "grad_norm": 0.38668930530548096, "learning_rate": 1e-06, "loss": 0.0099, "num_tokens": 602948623.0, "reward": 0.56640625, "reward_std": 0.21668873727321625, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1496, "tools/generated_tokens": 3732.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1075.54296875, "completions/mean_terminated_length": 1040.109375, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.13240213738754392, "epoch": 0.25509617227937886, "frac_reward_zero_std": 0.6875, "grad_norm": 0.259509414434433, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 603298858.0, "reward": 0.57421875, "reward_std": 0.09914018213748932, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1497, "tools/generated_tokens": 3707.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1196.4921875, "completions/mean_terminated_length": 1108.4051513671875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.12543504452332854, "epoch": 0.2552665772040812, "frac_reward_zero_std": 0.5, "grad_norm": 0.3310832381248474, "learning_rate": 1e-06, "loss": -0.0006, "num_tokens": 603689272.0, "reward": 0.3828125, "reward_std": 0.19617824256420135, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1498, "tools/generated_tokens": 4316.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1325.0859375, "completions/mean_terminated_length": 1162.5167236328125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11327816918492317, "epoch": 0.2554369821287835, "frac_reward_zero_std": 0.625, "grad_norm": 0.31929364800453186, "learning_rate": 1e-06, "loss": 0.0162, "num_tokens": 604105438.0, "reward": 0.46875, "reward_std": 0.15796709060668945, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1499, "tools/generated_tokens": 5157.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1137.0078125, "completions/mean_terminated_length": 1029.5982666015625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.1506450902670622, "epoch": 0.25560738705348585, "frac_reward_zero_std": 0.5, "grad_norm": 0.36400106549263, "learning_rate": 1e-06, "loss": -0.0308, "num_tokens": 604480640.0, "reward": 0.34765625, "reward_std": 0.21213586628437042, "rewards/simpleverify_reward/mean": 0.34765625, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1500, "tools/generated_tokens": 4649.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1065.92578125, "completions/mean_terminated_length": 1021.8325805664062, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.1170584256760776, "epoch": 0.2557777919781882, "frac_reward_zero_std": 0.4375, "grad_norm": 1.011528730392456, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 604820333.0, "reward": 0.72265625, "reward_std": 0.24993236362934113, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1501, "tools/generated_tokens": 3537.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1132.00390625, "completions/mean_terminated_length": 1041.5880126953125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.11548105161637068, "epoch": 0.2559481969028905, "frac_reward_zero_std": 0.5625, "grad_norm": 0.36953091621398926, "learning_rate": 1e-06, "loss": 0.0248, "num_tokens": 605171598.0, "reward": 0.70703125, "reward_std": 0.14604227244853973, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 1502, "tools/generated_tokens": 3252.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.03515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1071.03515625, "completions/mean_terminated_length": 992.7130126953125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.14151150919497013, "epoch": 0.25611860182759283, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3572862148284912, "learning_rate": 1e-06, "loss": -0.0266, "num_tokens": 605527975.0, "reward": 0.42578125, "reward_std": 0.2317298948764801, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1503, "tools/generated_tokens": 4431.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1092.484375, "completions/mean_terminated_length": 1065.6224365234375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.12230291403830051, "epoch": 0.25628900675229516, "frac_reward_zero_std": 0.625, "grad_norm": 0.26486676931381226, "learning_rate": 1e-06, "loss": 0.0003, "num_tokens": 605868195.0, "reward": 0.47265625, "reward_std": 0.12082062661647797, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1504, "tools/generated_tokens": 2964.484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1183.984375, "completions/mean_terminated_length": 1110.7626953125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.12293351627886295, "epoch": 0.2564594116769975, "frac_reward_zero_std": 0.4375, "grad_norm": 0.41259703040122986, "learning_rate": 1e-06, "loss": -0.0269, "num_tokens": 606245775.0, "reward": 0.59765625, "reward_std": 0.23559054732322693, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1505, "tools/generated_tokens": 3703.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1146.15234375, "completions/mean_terminated_length": 1048.5498046875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.129985386505723, "epoch": 0.2566298166016998, "frac_reward_zero_std": 0.5625, "grad_norm": 0.29228994250297546, "learning_rate": 1e-06, "loss": -0.0094, "num_tokens": 606613878.0, "reward": 0.5859375, "reward_std": 0.18763354420661926, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1506, "tools/generated_tokens": 4082.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1202.78125, "completions/mean_terminated_length": 1012.712890625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.11632753303274512, "epoch": 0.25680022152640214, "frac_reward_zero_std": 0.5, "grad_norm": 0.36938849091529846, "learning_rate": 1e-06, "loss": 0.0382, "num_tokens": 607001646.0, "reward": 0.640625, "reward_std": 0.1892854869365692, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1507, "tools/generated_tokens": 4466.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1298.359375, "completions/mean_terminated_length": 1163.635986328125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.11639787908643484, "epoch": 0.2569706264511044, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2725435495376587, "learning_rate": 1e-06, "loss": -0.009, "num_tokens": 607403434.0, "reward": 0.41796875, "reward_std": 0.16518138349056244, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1508, "tools/generated_tokens": 4386.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1136.390625, "completions/mean_terminated_length": 1075.61669921875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.12626192392781377, "epoch": 0.25714103137580674, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3339533507823944, "learning_rate": 1e-06, "loss": 0.0242, "num_tokens": 607760142.0, "reward": 0.4765625, "reward_std": 0.18159392476081848, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1509, "tools/generated_tokens": 3104.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1158.80859375, "completions/mean_terminated_length": 1107.371826171875, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.12791414512321353, "epoch": 0.25731143630050907, "frac_reward_zero_std": 0.4375, "grad_norm": 0.596592128276825, "learning_rate": 1e-06, "loss": 0.0143, "num_tokens": 608129789.0, "reward": 0.5703125, "reward_std": 0.22138670086860657, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1510, "tools/generated_tokens": 3862.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1245.9296875, "completions/mean_terminated_length": 1097.398193359375, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.10400097258388996, "epoch": 0.2574818412252114, "frac_reward_zero_std": 0.5625, "grad_norm": 0.34309878945350647, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 608516363.0, "reward": 0.625, "reward_std": 0.13423693180084229, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1511, "tools/generated_tokens": 4021.921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.35546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1189.7734375, "completions/mean_terminated_length": 1120.970458984375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.1399524286389351, "epoch": 0.2576522461499137, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3278346359729767, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 608900705.0, "reward": 0.6171875, "reward_std": 0.14954319596290588, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1512, "tools/generated_tokens": 4093.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1210.24609375, "completions/mean_terminated_length": 1179.720703125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.14155074208974838, "epoch": 0.25782265107461605, "frac_reward_zero_std": 0.5, "grad_norm": 0.5441141128540039, "learning_rate": 1e-06, "loss": 0.0337, "num_tokens": 609273072.0, "reward": 0.7265625, "reward_std": 0.17254294455051422, "rewards/simpleverify_reward/mean": 0.7265625, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 1513, "tools/generated_tokens": 3386.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1171.09375, "completions/mean_terminated_length": 1108.7196044921875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.1335051991045475, "epoch": 0.2579930559993184, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3972628712654114, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 609645224.0, "reward": 0.65625, "reward_std": 0.15911275148391724, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1514, "tools/generated_tokens": 3923.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1131.703125, "completions/mean_terminated_length": 1041.253173828125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.1350549110211432, "epoch": 0.2581634609240207, "frac_reward_zero_std": 0.625, "grad_norm": 0.3876400589942932, "learning_rate": 1e-06, "loss": 0.019, "num_tokens": 610003324.0, "reward": 0.5234375, "reward_std": 0.1504095196723938, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1515, "tools/generated_tokens": 3675.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1108.65625, "completions/mean_terminated_length": 1050.195068359375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.13995753787457943, "epoch": 0.25833386584872303, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3055340349674225, "learning_rate": 1e-06, "loss": 0.0156, "num_tokens": 610360564.0, "reward": 0.57421875, "reward_std": 0.20223368704319, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1516, "tools/generated_tokens": 3876.66796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1286.890625, "completions/mean_terminated_length": 1120.2000732421875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.1227824641391635, "epoch": 0.25850427077342536, "frac_reward_zero_std": 0.375, "grad_norm": 0.44475123286247253, "learning_rate": 1e-06, "loss": 0.0253, "num_tokens": 610766264.0, "reward": 0.44921875, "reward_std": 0.22500738501548767, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1517, "tools/generated_tokens": 5134.91796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1139.828125, "completions/mean_terminated_length": 1075.2301025390625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.13427804876118898, "epoch": 0.2586746756981277, "frac_reward_zero_std": 0.5, "grad_norm": 0.3722994327545166, "learning_rate": 1e-06, "loss": 0.0067, "num_tokens": 611145212.0, "reward": 0.609375, "reward_std": 0.1737399697303772, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1518, "tools/generated_tokens": 4147.8203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1223.2421875, "completions/mean_terminated_length": 1117.8765869140625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.14036784507334232, "epoch": 0.25884508062283, "frac_reward_zero_std": 0.6875, "grad_norm": 0.21806803345680237, "learning_rate": 1e-06, "loss": -0.0037, "num_tokens": 611530794.0, "reward": 0.33984375, "reward_std": 0.09947281330823898, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1519, "tools/generated_tokens": 4207.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2009.0, "completions/mean_length": 1183.96484375, "completions/mean_terminated_length": 1106.753173828125, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.1300795474089682, "epoch": 0.25901548554753234, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3282361626625061, "learning_rate": 1e-06, "loss": 0.0115, "num_tokens": 611907473.0, "reward": 0.67578125, "reward_std": 0.20873014628887177, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1520, "tools/generated_tokens": 3943.9609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1173.6953125, "completions/mean_terminated_length": 1083.25, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.13029592065140605, "epoch": 0.25918589047223467, "frac_reward_zero_std": 0.4375, "grad_norm": 0.40472638607025146, "learning_rate": 1e-06, "loss": 0.0317, "num_tokens": 612281747.0, "reward": 0.5625, "reward_std": 0.23528026044368744, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1521, "tools/generated_tokens": 4389.7109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1248.296875, "completions/mean_terminated_length": 1154.0218505859375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.12178336596116424, "epoch": 0.259356295396937, "frac_reward_zero_std": 0.5, "grad_norm": 0.3459000885486603, "learning_rate": 1e-06, "loss": 0.0184, "num_tokens": 612678607.0, "reward": 0.29296875, "reward_std": 0.19672392308712006, "rewards/simpleverify_reward/mean": 0.29296875, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 1522, "tools/generated_tokens": 4856.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1163.09375, "completions/mean_terminated_length": 1058.7598876953125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.11637395201250911, "epoch": 0.25952670032163927, "frac_reward_zero_std": 0.4375, "grad_norm": 0.39401382207870483, "learning_rate": 1e-06, "loss": -0.0091, "num_tokens": 613055895.0, "reward": 0.3359375, "reward_std": 0.21047285199165344, "rewards/simpleverify_reward/mean": 0.3359375, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 1523, "tools/generated_tokens": 4875.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.8125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1079.4296875, "completions/mean_terminated_length": 1031.7950439453125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.11004726868122816, "epoch": 0.2596971052463416, "frac_reward_zero_std": 0.5, "grad_norm": 0.374600350856781, "learning_rate": 1e-06, "loss": -0.0202, "num_tokens": 613401445.0, "reward": 0.55078125, "reward_std": 0.19785727560520172, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1524, "tools/generated_tokens": 4455.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1184.29296875, "completions/mean_terminated_length": 1056.4798583984375, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.11089019337669015, "epoch": 0.2598675101710439, "frac_reward_zero_std": 0.4375, "grad_norm": 0.29083749651908875, "learning_rate": 1e-06, "loss": 0.0418, "num_tokens": 613782000.0, "reward": 0.4609375, "reward_std": 0.20368444919586182, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1525, "tools/generated_tokens": 4456.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 1144.796875, "completions/mean_terminated_length": 1111.88671875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.14034538716077805, "epoch": 0.26003791509574625, "frac_reward_zero_std": 0.75, "grad_norm": 0.2658088803291321, "learning_rate": 1e-06, "loss": -0.0227, "num_tokens": 614140476.0, "reward": 0.6171875, "reward_std": 0.11533986032009125, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1526, "tools/generated_tokens": 3296.796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1118.19921875, "completions/mean_terminated_length": 1056.2125244140625, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.10838120291009545, "epoch": 0.2602083200204486, "frac_reward_zero_std": 0.375, "grad_norm": 0.3732198476791382, "learning_rate": 1e-06, "loss": 0.0057, "num_tokens": 614490943.0, "reward": 0.65625, "reward_std": 0.2147856056690216, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1527, "tools/generated_tokens": 3782.19140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1203.47265625, "completions/mean_terminated_length": 1078.497802734375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.13018015585839748, "epoch": 0.2603787249451509, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3032432496547699, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 614867176.0, "reward": 0.51953125, "reward_std": 0.1468954086303711, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1528, "tools/generated_tokens": 3755.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1268.47265625, "completions/mean_terminated_length": 1115.4813232421875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.1120557775720954, "epoch": 0.26054912986985324, "frac_reward_zero_std": 0.375, "grad_norm": 0.5772393345832825, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 615272257.0, "reward": 0.2890625, "reward_std": 0.2180875539779663, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1529, "tools/generated_tokens": 4684.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1208.640625, "completions/mean_terminated_length": 1133.634033203125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.11497182631865144, "epoch": 0.26071953479455556, "frac_reward_zero_std": 0.5, "grad_norm": 0.32598379254341125, "learning_rate": 1e-06, "loss": -0.0323, "num_tokens": 615652565.0, "reward": 0.4296875, "reward_std": 0.17829003930091858, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1530, "tools/generated_tokens": 4000.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1126.59765625, "completions/mean_terminated_length": 1081.28271484375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.11353438859805465, "epoch": 0.2608899397192579, "frac_reward_zero_std": 0.625, "grad_norm": 0.3886694312095642, "learning_rate": 1e-06, "loss": 0.0444, "num_tokens": 616013166.0, "reward": 0.4375, "reward_std": 0.13149453699588776, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1531, "tools/generated_tokens": 3742.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1050.79296875, "completions/mean_terminated_length": 1030.9283447265625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.12735770735889673, "epoch": 0.2610603446439602, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4155341684818268, "learning_rate": 1e-06, "loss": 0.0328, "num_tokens": 616354313.0, "reward": 0.43359375, "reward_std": 0.1845875382423401, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1532, "tools/generated_tokens": 3674.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1070.1953125, "completions/mean_terminated_length": 1062.49609375, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.12446552701294422, "epoch": 0.26123074956866255, "frac_reward_zero_std": 0.5625, "grad_norm": 0.47665664553642273, "learning_rate": 1e-06, "loss": -0.0119, "num_tokens": 616693867.0, "reward": 0.58203125, "reward_std": 0.15613234043121338, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1533, "tools/generated_tokens": 2950.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1264.80859375, "completions/mean_terminated_length": 1050.5074462890625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.11805044766515493, "epoch": 0.2614011544933649, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3420359492301941, "learning_rate": 1e-06, "loss": -0.0115, "num_tokens": 617082874.0, "reward": 0.46875, "reward_std": 0.16924381256103516, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1534, "tools/generated_tokens": 4352.82421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1268.11328125, "completions/mean_terminated_length": 1156.700927734375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.11775952018797398, "epoch": 0.2615715594180672, "frac_reward_zero_std": 0.375, "grad_norm": 0.5181947946548462, "learning_rate": 1e-06, "loss": -0.0005, "num_tokens": 617479255.0, "reward": 0.61328125, "reward_std": 0.24988514184951782, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1535, "tools/generated_tokens": 4236.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1237.66015625, "completions/mean_terminated_length": 1092.0230712890625, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.11667987145483494, "epoch": 0.26174196434276953, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3779032528400421, "learning_rate": 1e-06, "loss": 0.0346, "num_tokens": 617878944.0, "reward": 0.47265625, "reward_std": 0.2455040067434311, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1536, "tools/generated_tokens": 4805.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1234.44140625, "completions/mean_terminated_length": 1142.473876953125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.11061756033450365, "epoch": 0.26191236926747186, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4314779043197632, "learning_rate": 1e-06, "loss": -0.0003, "num_tokens": 618272881.0, "reward": 0.546875, "reward_std": 0.2260051667690277, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1537, "tools/generated_tokens": 4434.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1117.40234375, "completions/mean_terminated_length": 1067.6173095703125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.10649182088673115, "epoch": 0.26208277419217413, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2916627824306488, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 618632536.0, "reward": 0.48828125, "reward_std": 0.13528887927532196, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1538, "tools/generated_tokens": 3773.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1247.5625, "completions/mean_terminated_length": 1120.81005859375, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.1157947750762105, "epoch": 0.26225317911687646, "frac_reward_zero_std": 0.375, "grad_norm": 0.45595791935920715, "learning_rate": 1e-06, "loss": 0.0006, "num_tokens": 619046904.0, "reward": 0.37109375, "reward_std": 0.24350816011428833, "rewards/simpleverify_reward/mean": 0.37109375, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1539, "tools/generated_tokens": 4807.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1073.94921875, "completions/mean_terminated_length": 1034.3536376953125, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.11641758494079113, "epoch": 0.2624235840415788, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3148609399795532, "learning_rate": 1e-06, "loss": 0.0006, "num_tokens": 619389307.0, "reward": 0.5546875, "reward_std": 0.1624118983745575, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1540, "tools/generated_tokens": 3425.95703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1149.6328125, "completions/mean_terminated_length": 1116.8988037109375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.11827847454696894, "epoch": 0.2625939889662811, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3843304514884949, "learning_rate": 1e-06, "loss": 0.0127, "num_tokens": 619757581.0, "reward": 0.3984375, "reward_std": 0.2079564929008484, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1541, "tools/generated_tokens": 3741.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1045.3671875, "completions/mean_terminated_length": 1013.024169921875, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.1081991302780807, "epoch": 0.26276439389098344, "frac_reward_zero_std": 0.625, "grad_norm": 0.4231826364994049, "learning_rate": 1e-06, "loss": 0.0155, "num_tokens": 620095419.0, "reward": 0.484375, "reward_std": 0.15671618282794952, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1542, "tools/generated_tokens": 3893.3671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1091.1796875, "completions/mean_terminated_length": 1075.9920654296875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11913190595805645, "epoch": 0.26293479881568577, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3482901453971863, "learning_rate": 1e-06, "loss": -0.0245, "num_tokens": 620446601.0, "reward": 0.57421875, "reward_std": 0.21777918934822083, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1543, "tools/generated_tokens": 3731.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1201.7265625, "completions/mean_terminated_length": 1122.183837890625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11702842870727181, "epoch": 0.2631052037403881, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3938402235507965, "learning_rate": 1e-06, "loss": 0.0007, "num_tokens": 620828147.0, "reward": 0.359375, "reward_std": 0.24525083601474762, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1544, "tools/generated_tokens": 4097.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1113.98046875, "completions/mean_terminated_length": 999.2763061523438, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.1085619698278606, "epoch": 0.2632756086650904, "frac_reward_zero_std": 0.6875, "grad_norm": 0.27679890394210815, "learning_rate": 1e-06, "loss": 0.0194, "num_tokens": 621189662.0, "reward": 0.34375, "reward_std": 0.10958996415138245, "rewards/simpleverify_reward/mean": 0.34375, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1545, "tools/generated_tokens": 4377.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1184.42578125, "completions/mean_terminated_length": 1160.152587890625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11222513997927308, "epoch": 0.26344601358979275, "frac_reward_zero_std": 0.5, "grad_norm": 0.3617621064186096, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 621566939.0, "reward": 0.56640625, "reward_std": 0.18903234601020813, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1546, "tools/generated_tokens": 3720.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1173.625, "completions/mean_terminated_length": 1099.525390625, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.11120434105396271, "epoch": 0.2636164185144951, "frac_reward_zero_std": 0.4375, "grad_norm": 0.45923715829849243, "learning_rate": 1e-06, "loss": 0.0255, "num_tokens": 621943899.0, "reward": 0.4140625, "reward_std": 0.2565644085407257, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1547, "tools/generated_tokens": 4333.625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.54296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1131.11328125, "completions/mean_terminated_length": 1057.6075439453125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.11167830182239413, "epoch": 0.2637868234391974, "frac_reward_zero_std": 0.5, "grad_norm": 0.4154977798461914, "learning_rate": 1e-06, "loss": 0.0247, "num_tokens": 622313576.0, "reward": 0.71484375, "reward_std": 0.19146710634231567, "rewards/simpleverify_reward/mean": 0.71484375, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 1548, "tools/generated_tokens": 4179.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1116.95703125, "completions/mean_terminated_length": 1059.00830078125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.11047542607411742, "epoch": 0.26395722836389973, "frac_reward_zero_std": 0.5, "grad_norm": 0.3766826093196869, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 622679757.0, "reward": 0.5546875, "reward_std": 0.19607165455818176, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1549, "tools/generated_tokens": 4116.953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1205.78125, "completions/mean_terminated_length": 1167.96728515625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.12424036161974072, "epoch": 0.26412763328860206, "frac_reward_zero_std": 0.625, "grad_norm": 0.24687644839286804, "learning_rate": 1e-06, "loss": -0.0182, "num_tokens": 623060837.0, "reward": 0.47265625, "reward_std": 0.14711953699588776, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1550, "tools/generated_tokens": 3757.7890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1109.046875, "completions/mean_terminated_length": 1011.913818359375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.10488031525164843, "epoch": 0.2642980382133044, "frac_reward_zero_std": 0.5625, "grad_norm": 0.44755178689956665, "learning_rate": 1e-06, "loss": 0.0023, "num_tokens": 623418097.0, "reward": 0.5234375, "reward_std": 0.19274510443210602, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1551, "tools/generated_tokens": 3901.0546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1150.51953125, "completions/mean_terminated_length": 1082.6429443359375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.11244284873828292, "epoch": 0.2644684431380067, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4498797357082367, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 623783382.0, "reward": 0.484375, "reward_std": 0.23635753989219666, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1552, "tools/generated_tokens": 3870.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1037.1640625, "completions/mean_terminated_length": 946.8382568359375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.10186654608696699, "epoch": 0.264638848062709, "frac_reward_zero_std": 0.375, "grad_norm": 0.42248308658599854, "learning_rate": 1e-06, "loss": 0.0038, "num_tokens": 624121824.0, "reward": 0.53515625, "reward_std": 0.24943023920059204, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1553, "tools/generated_tokens": 3797.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1260.98828125, "completions/mean_terminated_length": 1212.0042724609375, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.11327718757092953, "epoch": 0.2648092529874113, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3770925998687744, "learning_rate": 1e-06, "loss": 0.0282, "num_tokens": 624500605.0, "reward": 0.69921875, "reward_std": 0.1921054571866989, "rewards/simpleverify_reward/mean": 0.69921875, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 1554, "tools/generated_tokens": 3052.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1150.56640625, "completions/mean_terminated_length": 1082.693359375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.11488336976617575, "epoch": 0.26497965791211364, "frac_reward_zero_std": 0.6875, "grad_norm": 0.32059574127197266, "learning_rate": 1e-06, "loss": 0.0227, "num_tokens": 624877742.0, "reward": 0.4375, "reward_std": 0.12860959768295288, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1555, "tools/generated_tokens": 3414.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1180.5390625, "completions/mean_terminated_length": 1033.981689453125, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.0921782380901277, "epoch": 0.26515006283681597, "frac_reward_zero_std": 0.5625, "grad_norm": 0.32096394896507263, "learning_rate": 1e-06, "loss": 0.0365, "num_tokens": 625250696.0, "reward": 0.53125, "reward_std": 0.184334397315979, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1556, "tools/generated_tokens": 4156.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1169.84375, "completions/mean_terminated_length": 1044.3929443359375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.11326998705044389, "epoch": 0.2653204677615183, "frac_reward_zero_std": 0.625, "grad_norm": 0.38056260347366333, "learning_rate": 1e-06, "loss": 0.0232, "num_tokens": 625616064.0, "reward": 0.52734375, "reward_std": 0.15031491219997406, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1557, "tools/generated_tokens": 3681.8515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1140.29296875, "completions/mean_terminated_length": 1091.732421875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.10364710306748748, "epoch": 0.2654908726862206, "frac_reward_zero_std": 0.6875, "grad_norm": 0.27103686332702637, "learning_rate": 1e-06, "loss": 0.0038, "num_tokens": 625972795.0, "reward": 0.55859375, "reward_std": 0.13391819596290588, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1558, "tools/generated_tokens": 3300.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1137.5390625, "completions/mean_terminated_length": 1100.5284423828125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.10120109282433987, "epoch": 0.26566127761092295, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2961196303367615, "learning_rate": 1e-06, "loss": 0.026, "num_tokens": 626321189.0, "reward": 0.63671875, "reward_std": 0.18793907761573792, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1559, "tools/generated_tokens": 2913.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1173.93359375, "completions/mean_terminated_length": 1091.7564697265625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.09794620331376791, "epoch": 0.2658316825356253, "frac_reward_zero_std": 0.375, "grad_norm": 0.3517288863658905, "learning_rate": 1e-06, "loss": -0.0291, "num_tokens": 626688532.0, "reward": 0.50390625, "reward_std": 0.22005629539489746, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1560, "tools/generated_tokens": 3469.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2010.0, "completions/mean_length": 1109.2265625, "completions/mean_terminated_length": 1078.9434814453125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.09715089108794928, "epoch": 0.2660020874603276, "frac_reward_zero_std": 0.375, "grad_norm": 0.5311896204948425, "learning_rate": 1e-06, "loss": -0.0422, "num_tokens": 627029534.0, "reward": 0.578125, "reward_std": 0.24688635766506195, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1561, "tools/generated_tokens": 3045.21875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1135.546875, "completions/mean_terminated_length": 1000.5247192382812, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.10837089503183961, "epoch": 0.26617249238502994, "frac_reward_zero_std": 0.5625, "grad_norm": 0.42574259638786316, "learning_rate": 1e-06, "loss": -0.0121, "num_tokens": 627408154.0, "reward": 0.4140625, "reward_std": 0.15614622831344604, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1562, "tools/generated_tokens": 4343.55078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.56640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11328125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1160.34375, "completions/mean_terminated_length": 1046.9427490234375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11549530318006873, "epoch": 0.26634289730973226, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3744671642780304, "learning_rate": 1e-06, "loss": -0.0238, "num_tokens": 627784786.0, "reward": 0.3828125, "reward_std": 0.1624118983745575, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1563, "tools/generated_tokens": 3992.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1070.3515625, "completions/mean_terminated_length": 1066.5177001953125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.13392597995698452, "epoch": 0.2665133022344346, "frac_reward_zero_std": 0.5, "grad_norm": 0.32552748918533325, "learning_rate": 1e-06, "loss": -0.0519, "num_tokens": 628133516.0, "reward": 0.7421875, "reward_std": 0.17099951207637787, "rewards/simpleverify_reward/mean": 0.7421875, "rewards/simpleverify_reward/std": 0.4382871091365814, "step": 1564, "tools/generated_tokens": 3062.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1050.5078125, "completions/mean_terminated_length": 956.7265625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.11850820435211062, "epoch": 0.2666837071591369, "frac_reward_zero_std": 0.3125, "grad_norm": 0.9672841429710388, "learning_rate": 1e-06, "loss": 0.0307, "num_tokens": 628478494.0, "reward": 0.59765625, "reward_std": 0.27179163694381714, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1565, "tools/generated_tokens": 3714.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1200.1328125, "completions/mean_terminated_length": 1172.7822265625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.10558189265429974, "epoch": 0.26685411208383925, "frac_reward_zero_std": 0.5, "grad_norm": 0.3423413634300232, "learning_rate": 1e-06, "loss": 0.05, "num_tokens": 628834240.0, "reward": 0.65234375, "reward_std": 0.22415617108345032, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1566, "tools/generated_tokens": 2424.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1085.27734375, "completions/mean_terminated_length": 1042.052978515625, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.12675105314701796, "epoch": 0.2670245170085416, "frac_reward_zero_std": 0.5, "grad_norm": 0.442286878824234, "learning_rate": 1e-06, "loss": 0.0148, "num_tokens": 629195399.0, "reward": 0.453125, "reward_std": 0.22485950589179993, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1567, "tools/generated_tokens": 4509.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1135.02734375, "completions/mean_terminated_length": 1040.586181640625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.10887016588822007, "epoch": 0.26719492193324385, "frac_reward_zero_std": 0.5, "grad_norm": 0.4803633987903595, "learning_rate": 1e-06, "loss": 0.023, "num_tokens": 629566366.0, "reward": 0.39453125, "reward_std": 0.20167499780654907, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1568, "tools/generated_tokens": 4167.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1052.7421875, "completions/mean_terminated_length": 1040.9407958984375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.11663123359903693, "epoch": 0.2673653268579462, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2211681455373764, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 629896764.0, "reward": 0.60546875, "reward_std": 0.09617365896701813, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1569, "tools/generated_tokens": 2844.7421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1095.92578125, "completions/mean_terminated_length": 1065.213623046875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.12110176961869001, "epoch": 0.2675357317826485, "frac_reward_zero_std": 0.5, "grad_norm": 0.44080743193626404, "learning_rate": 1e-06, "loss": 0.0137, "num_tokens": 630247609.0, "reward": 0.60546875, "reward_std": 0.2014508694410324, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1570, "tools/generated_tokens": 3319.92578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1190.24609375, "completions/mean_terminated_length": 1140.6239013671875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.12540008034557104, "epoch": 0.26770613670735083, "frac_reward_zero_std": 0.5, "grad_norm": 0.34095123410224915, "learning_rate": 1e-06, "loss": -0.0276, "num_tokens": 630634248.0, "reward": 0.3515625, "reward_std": 0.1786910742521286, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1571, "tools/generated_tokens": 3798.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1156.125, "completions/mean_terminated_length": 1112.26220703125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.14124542940407991, "epoch": 0.26787654163205316, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4057638943195343, "learning_rate": 1e-06, "loss": 0.0303, "num_tokens": 631004328.0, "reward": 0.48828125, "reward_std": 0.2893853783607483, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1572, "tools/generated_tokens": 4068.125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1099.765625, "completions/mean_terminated_length": 1096.047119140625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.11976015288382769, "epoch": 0.2680469465567555, "frac_reward_zero_std": 0.5625, "grad_norm": 0.28377464413642883, "learning_rate": 1e-06, "loss": 0.0075, "num_tokens": 631343836.0, "reward": 0.53125, "reward_std": 0.1544942855834961, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1573, "tools/generated_tokens": 2819.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.83984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2005.0, "completions/mean_length": 963.8125, "completions/mean_terminated_length": 937.7920532226562, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.13594621792435646, "epoch": 0.2682173514814578, "frac_reward_zero_std": 0.375, "grad_norm": 0.3722911477088928, "learning_rate": 1e-06, "loss": -0.0158, "num_tokens": 631676364.0, "reward": 0.45703125, "reward_std": 0.2476080358028412, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1574, "tools/generated_tokens": 3603.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1084.0390625, "completions/mean_terminated_length": 1068.7381591796875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11019337736070156, "epoch": 0.26838775640616014, "frac_reward_zero_std": 0.8125, "grad_norm": 0.19260449707508087, "learning_rate": 1e-06, "loss": -0.0291, "num_tokens": 632016774.0, "reward": 0.4296875, "reward_std": 0.091475710272789, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1575, "tools/generated_tokens": 3444.05859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1030.07421875, "completions/mean_terminated_length": 1022.0590209960938, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.11183864856138825, "epoch": 0.26855816133086247, "frac_reward_zero_std": 0.5, "grad_norm": 0.5350489616394043, "learning_rate": 1e-06, "loss": -0.0197, "num_tokens": 632350521.0, "reward": 0.578125, "reward_std": 0.19475515186786652, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1576, "tools/generated_tokens": 3534.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1020.8828125, "completions/mean_terminated_length": 1000.42236328125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.11673133261501789, "epoch": 0.2687285662555648, "frac_reward_zero_std": 0.5, "grad_norm": 0.36078548431396484, "learning_rate": 1e-06, "loss": -0.0045, "num_tokens": 632692427.0, "reward": 0.5078125, "reward_std": 0.19258464872837067, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1577, "tools/generated_tokens": 3532.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1196.43359375, "completions/mean_terminated_length": 1154.55322265625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.10627295775339007, "epoch": 0.2688989711802671, "frac_reward_zero_std": 0.4375, "grad_norm": 0.46550074219703674, "learning_rate": 1e-06, "loss": -0.0132, "num_tokens": 633063338.0, "reward": 0.640625, "reward_std": 0.20778432488441467, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1578, "tools/generated_tokens": 3148.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1172.89453125, "completions/mean_terminated_length": 1155.462158203125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.12410266604274511, "epoch": 0.26906937610496945, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2918483018875122, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 633421551.0, "reward": 0.4765625, "reward_std": 0.13713786005973816, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1579, "tools/generated_tokens": 2420.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1089.7734375, "completions/mean_terminated_length": 1050.821044921875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.11814267467707396, "epoch": 0.2692397810296718, "frac_reward_zero_std": 0.375, "grad_norm": 0.40293872356414795, "learning_rate": 1e-06, "loss": 0.0107, "num_tokens": 633767157.0, "reward": 0.5546875, "reward_std": 0.25002697110176086, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1580, "tools/generated_tokens": 3001.7734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 919.48828125, "completions/mean_terminated_length": 915.0628051757812, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.11299776704981923, "epoch": 0.2694101859543741, "frac_reward_zero_std": 0.5, "grad_norm": 0.4657357335090637, "learning_rate": 1e-06, "loss": -0.0503, "num_tokens": 634059314.0, "reward": 0.3828125, "reward_std": 0.20112933218479156, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1581, "tools/generated_tokens": 2407.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1087.73828125, "completions/mean_terminated_length": 1060.742919921875, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.1092245695181191, "epoch": 0.26958059087907643, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5727904438972473, "learning_rate": 1e-06, "loss": -0.007, "num_tokens": 634403679.0, "reward": 0.5390625, "reward_std": 0.17693254351615906, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1582, "tools/generated_tokens": 3175.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 949.65234375, "completions/mean_terminated_length": 945.3451538085938, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.11242018034681678, "epoch": 0.2697509958037787, "frac_reward_zero_std": 0.5625, "grad_norm": 0.32268026471138, "learning_rate": 1e-06, "loss": -0.0117, "num_tokens": 634711734.0, "reward": 0.6015625, "reward_std": 0.16998592019081116, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1583, "tools/generated_tokens": 2733.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.87109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1094.30078125, "completions/mean_terminated_length": 1063.5362548828125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.10976075939834118, "epoch": 0.26992140072848103, "frac_reward_zero_std": 0.625, "grad_norm": 0.29567354917526245, "learning_rate": 1e-06, "loss": -0.0026, "num_tokens": 635057219.0, "reward": 0.41015625, "reward_std": 0.13505156338214874, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1584, "tools/generated_tokens": 3150.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.00390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1138.23046875, "completions/mean_terminated_length": 1131.06689453125, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.10552078438922763, "epoch": 0.27009180565318336, "frac_reward_zero_std": 0.375, "grad_norm": 0.3967992067337036, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 635408878.0, "reward": 0.53125, "reward_std": 0.29003632068634033, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1585, "tools/generated_tokens": 2970.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.89453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1012.375, "completions/mean_terminated_length": 1008.3137817382812, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.11210188921540976, "epoch": 0.2702622105778857, "frac_reward_zero_std": 0.75, "grad_norm": 0.24448874592781067, "learning_rate": 1e-06, "loss": -0.0136, "num_tokens": 635717694.0, "reward": 0.453125, "reward_std": 0.07206955552101135, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1586, "tools/generated_tokens": 2540.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.74609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1095.11328125, "completions/mean_terminated_length": 1072.2440185546875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12204699078574777, "epoch": 0.270432615502588, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4413423240184784, "learning_rate": 1e-06, "loss": 0.0067, "num_tokens": 636059483.0, "reward": 0.65625, "reward_std": 0.21205638349056244, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1587, "tools/generated_tokens": 2703.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.78515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1100.1640625, "completions/mean_terminated_length": 1057.608154296875, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.09738766215741634, "epoch": 0.27060302042729034, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4475040137767792, "learning_rate": 1e-06, "loss": 0.0287, "num_tokens": 636402485.0, "reward": 0.66796875, "reward_std": 0.16569995880126953, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1588, "tools/generated_tokens": 2812.15625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1079.4375, "completions/mean_terminated_length": 1060.1434326171875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.11382754379883409, "epoch": 0.27077342535199267, "frac_reward_zero_std": 0.625, "grad_norm": 0.40638142824172974, "learning_rate": 1e-06, "loss": -0.0326, "num_tokens": 636728117.0, "reward": 0.44921875, "reward_std": 0.14383356273174286, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1589, "tools/generated_tokens": 2759.4375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1065.71875, "completions/mean_terminated_length": 1054.0711669921875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10323592089116573, "epoch": 0.270943830276695, "frac_reward_zero_std": 0.5, "grad_norm": 0.4680120050907135, "learning_rate": 1e-06, "loss": 0.0277, "num_tokens": 637076573.0, "reward": 0.59375, "reward_std": 0.19970625638961792, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1590, "tools/generated_tokens": 2921.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.90625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2035.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 959.0078125, "completions/mean_terminated_length": 959.0078125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.09263528464362025, "epoch": 0.2711142352013973, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4703959822654724, "learning_rate": 1e-06, "loss": -0.0576, "num_tokens": 637389215.0, "reward": 0.69140625, "reward_std": 0.2617402672767639, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1591, "tools/generated_tokens": 2847.0078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2041.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1024.63671875, "completions/mean_terminated_length": 1024.63671875, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.10179853392764926, "epoch": 0.27128464012609965, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3974047601222992, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 637727442.0, "reward": 0.5859375, "reward_std": 0.16769562661647797, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1592, "tools/generated_tokens": 3288.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 1988.0, "completions/max_terminated_length": 1988.0, "completions/mean_length": 996.11328125, "completions/mean_terminated_length": 996.11328125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.08716936549171805, "epoch": 0.271455045050802, "frac_reward_zero_std": 0.5, "grad_norm": 0.3834003210067749, "learning_rate": 1e-06, "loss": -0.0064, "num_tokens": 638036351.0, "reward": 0.83203125, "reward_std": 0.1780368983745575, "rewards/simpleverify_reward/mean": 0.83203125, "rewards/simpleverify_reward/std": 0.3745708465576172, "step": 1593, "tools/generated_tokens": 1988.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 957.5078125, "completions/mean_terminated_length": 953.2314453125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.09403038769960403, "epoch": 0.2716254499755043, "frac_reward_zero_std": 0.3125, "grad_norm": 0.7058821320533752, "learning_rate": 1e-06, "loss": -0.0022, "num_tokens": 638356321.0, "reward": 0.703125, "reward_std": 0.2325398027896881, "rewards/simpleverify_reward/mean": 0.703125, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 1594, "tools/generated_tokens": 3141.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2032.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 922.45703125, "completions/mean_terminated_length": 922.45703125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.11052666511386633, "epoch": 0.27179585490020663, "frac_reward_zero_std": 0.375, "grad_norm": 0.7321500778198242, "learning_rate": 1e-06, "loss": -0.0264, "num_tokens": 638671542.0, "reward": 0.45703125, "reward_std": 0.2199878990650177, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1595, "tools/generated_tokens": 3162.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1085.63671875, "completions/mean_terminated_length": 1042.4366455078125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.10116981761530042, "epoch": 0.27196625982490896, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6969082355499268, "learning_rate": 1e-06, "loss": -0.0285, "num_tokens": 639022649.0, "reward": 0.28515625, "reward_std": 0.20707818865776062, "rewards/simpleverify_reward/mean": 0.28515625, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 1596, "tools/generated_tokens": 4045.66796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2021.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1078.13671875, "completions/mean_terminated_length": 1078.13671875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.10450277803465724, "epoch": 0.2721366647496113, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3553864657878876, "learning_rate": 1e-06, "loss": 0.0241, "num_tokens": 639361324.0, "reward": 0.76171875, "reward_std": 0.13039492070674896, "rewards/simpleverify_reward/mean": 0.76171875, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 1597, "tools/generated_tokens": 2558.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.72265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1117.6796875, "completions/mean_terminated_length": 1114.031494140625, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.10889062844216824, "epoch": 0.27230706967431356, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4066812992095947, "learning_rate": 1e-06, "loss": -0.0097, "num_tokens": 639713146.0, "reward": 0.58984375, "reward_std": 0.17923866212368011, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1598, "tools/generated_tokens": 2821.67578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.83203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1109.6328125, "completions/mean_terminated_length": 1098.5059814453125, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.09621346229687333, "epoch": 0.2724774745990159, "frac_reward_zero_std": 0.375, "grad_norm": 0.9574671983718872, "learning_rate": 1e-06, "loss": -0.0019, "num_tokens": 640064444.0, "reward": 0.57421875, "reward_std": 0.24492931365966797, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1599, "tools/generated_tokens": 3221.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.03125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2040.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1008.76953125, "completions/mean_terminated_length": 1008.76953125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.09529742738232017, "epoch": 0.2726478795237182, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5640086531639099, "learning_rate": 1e-06, "loss": -0.0195, "num_tokens": 640379409.0, "reward": 0.578125, "reward_std": 0.17726992070674896, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1600, "tools/generated_tokens": 2416.76953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1054.2109375, "completions/mean_terminated_length": 1054.2109375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.09338672459125519, "epoch": 0.27281828444842054, "frac_reward_zero_std": 0.5625, "grad_norm": 0.43575453758239746, "learning_rate": 1e-06, "loss": 0.0104, "num_tokens": 640713063.0, "reward": 0.58984375, "reward_std": 0.16516819596290588, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1601, "tools/generated_tokens": 2654.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2033.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1107.953125, "completions/mean_terminated_length": 1107.953125, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.08641050988808274, "epoch": 0.2729886893731229, "frac_reward_zero_std": 0.5, "grad_norm": 0.4460669755935669, "learning_rate": 1e-06, "loss": -0.0471, "num_tokens": 641040203.0, "reward": 0.67578125, "reward_std": 0.17672231793403625, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1602, "tools/generated_tokens": 1995.9453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 972.39453125, "completions/mean_terminated_length": 972.39453125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10101018054410815, "epoch": 0.2731590942978252, "frac_reward_zero_std": 0.4375, "grad_norm": 0.435713529586792, "learning_rate": 1e-06, "loss": 0.0186, "num_tokens": 641346464.0, "reward": 0.62890625, "reward_std": 0.21999263763427734, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1603, "tools/generated_tokens": 2340.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1067.53515625, "completions/mean_terminated_length": 1051.9722900390625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.08552238391712308, "epoch": 0.2733294992225275, "frac_reward_zero_std": 0.1875, "grad_norm": 0.6458866596221924, "learning_rate": 1e-06, "loss": 0.0087, "num_tokens": 641689897.0, "reward": 0.4453125, "reward_std": 0.3158308267593384, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1604, "tools/generated_tokens": 3411.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2041.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1067.87109375, "completions/mean_terminated_length": 1067.87109375, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.08456784719601274, "epoch": 0.27349990414722986, "frac_reward_zero_std": 0.5, "grad_norm": 0.39203283190727234, "learning_rate": 1e-06, "loss": 0.039, "num_tokens": 642008936.0, "reward": 0.671875, "reward_std": 0.20811696350574493, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1605, "tools/generated_tokens": 2059.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1157.2421875, "completions/mean_terminated_length": 1153.7491455078125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.10706569720059633, "epoch": 0.2736703090719322, "frac_reward_zero_std": 0.5625, "grad_norm": 0.38836193084716797, "learning_rate": 1e-06, "loss": -0.019, "num_tokens": 642361638.0, "reward": 0.65625, "reward_std": 0.14737266302108765, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1606, "tools/generated_tokens": 2813.26171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.80859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 989.20703125, "completions/mean_terminated_length": 985.054931640625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.08730290457606316, "epoch": 0.2738407139966345, "frac_reward_zero_std": 0.4375, "grad_norm": 0.7079214453697205, "learning_rate": 1e-06, "loss": -0.045, "num_tokens": 642677899.0, "reward": 0.56640625, "reward_std": 0.18651601672172546, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1607, "tools/generated_tokens": 2925.2109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2002.0, "completions/mean_length": 1029.79296875, "completions/mean_terminated_length": 1005.3560180664062, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.08336546923965216, "epoch": 0.27401111892133684, "frac_reward_zero_std": 0.6875, "grad_norm": 0.340177983045578, "learning_rate": 1e-06, "loss": 0.0194, "num_tokens": 642987942.0, "reward": 0.40625, "reward_std": 0.14500631392002106, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1608, "tools/generated_tokens": 2285.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.61328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1075.04296875, "completions/mean_terminated_length": 1055.661376953125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.08411613944917917, "epoch": 0.27418152384603917, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4558895230293274, "learning_rate": 1e-06, "loss": -0.052, "num_tokens": 643319537.0, "reward": 0.66015625, "reward_std": 0.16619305312633514, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1609, "tools/generated_tokens": 2803.04296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2018.0, "completions/mean_length": 919.22265625, "completions/mean_terminated_length": 914.796142578125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.09673519060015678, "epoch": 0.2743519287707415, "frac_reward_zero_std": 0.6875, "grad_norm": 0.47374454140663147, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 643618490.0, "reward": 0.546875, "reward_std": 0.12388455867767334, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1610, "tools/generated_tokens": 2423.23828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1271.30859375, "completions/mean_terminated_length": 1216.062744140625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.08340617083013058, "epoch": 0.2745223336954438, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3683626055717468, "learning_rate": 1e-06, "loss": -0.0019, "num_tokens": 643996777.0, "reward": 0.3984375, "reward_std": 0.12939241528511047, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1611, "tools/generated_tokens": 3311.30078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1139.546875, "completions/mean_terminated_length": 1110.241943359375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08376920642331243, "epoch": 0.27469273862014615, "frac_reward_zero_std": 0.5, "grad_norm": 0.3009025752544403, "learning_rate": 1e-06, "loss": -0.0019, "num_tokens": 644341045.0, "reward": 0.421875, "reward_std": 0.1643964648246765, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1612, "tools/generated_tokens": 2699.546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2047.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1046.23046875, "completions/mean_terminated_length": 1042.305908203125, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.0879359096288681, "epoch": 0.2748631435448484, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5562757849693298, "learning_rate": 1e-06, "loss": -0.0391, "num_tokens": 644676016.0, "reward": 0.53125, "reward_std": 0.2706853747367859, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1613, "tools/generated_tokens": 2910.23828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.91015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1079.7421875, "completions/mean_terminated_length": 1075.9451904296875, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08398322854191065, "epoch": 0.27503354846955075, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2664090692996979, "learning_rate": 1e-06, "loss": -0.0237, "num_tokens": 645005342.0, "reward": 0.53125, "reward_std": 0.1304856687784195, "rewards/simpleverify_reward/mean": 0.53125, "rewards/simpleverify_reward/std": 0.5, "step": 1614, "tools/generated_tokens": 2439.73828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1055.41015625, "completions/mean_terminated_length": 1039.65478515625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.07957409741356969, "epoch": 0.2752039533942531, "frac_reward_zero_std": 0.625, "grad_norm": 0.3727455735206604, "learning_rate": 1e-06, "loss": 0.0045, "num_tokens": 645330951.0, "reward": 0.41015625, "reward_std": 0.14656084775924683, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1615, "tools/generated_tokens": 2591.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.75, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 957.24609375, "completions/mean_terminated_length": 952.9686889648438, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.0856359931640327, "epoch": 0.2753743583189554, "frac_reward_zero_std": 0.5, "grad_norm": 0.5019395351409912, "learning_rate": 1e-06, "loss": -0.0306, "num_tokens": 645651862.0, "reward": 0.58984375, "reward_std": 0.19125424325466156, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1616, "tools/generated_tokens": 3405.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1011.65234375, "completions/mean_terminated_length": 999.3636474609375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.09706755541265011, "epoch": 0.27554476324365773, "frac_reward_zero_std": 0.5, "grad_norm": 0.40986114740371704, "learning_rate": 1e-06, "loss": -0.0003, "num_tokens": 645967133.0, "reward": 0.47265625, "reward_std": 0.20992997288703918, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1617, "tools/generated_tokens": 2211.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.5859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1068.25390625, "completions/mean_terminated_length": 1048.737060546875, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.09371689986437559, "epoch": 0.27571516816836006, "frac_reward_zero_std": 0.5, "grad_norm": 0.5814691185951233, "learning_rate": 1e-06, "loss": 0.0282, "num_tokens": 646304654.0, "reward": 0.52734375, "reward_std": 0.24008628726005554, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1618, "tools/generated_tokens": 3108.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 983.19140625, "completions/mean_terminated_length": 953.2570190429688, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.08254634868353605, "epoch": 0.2758855730930624, "frac_reward_zero_std": 0.3125, "grad_norm": 0.6362841725349426, "learning_rate": 1e-06, "loss": 0.0229, "num_tokens": 646621231.0, "reward": 0.55859375, "reward_std": 0.29942113161087036, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1619, "tools/generated_tokens": 2831.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.90234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1026.68359375, "completions/mean_terminated_length": 1026.68359375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.08773822989314795, "epoch": 0.2760559780177647, "frac_reward_zero_std": 0.5, "grad_norm": 0.5758412480354309, "learning_rate": 1e-06, "loss": -0.0647, "num_tokens": 646950862.0, "reward": 0.3671875, "reward_std": 0.1892854869365692, "rewards/simpleverify_reward/mean": 0.3671875, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1620, "tools/generated_tokens": 2994.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1213.42578125, "completions/mean_terminated_length": 1172.3851318359375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.09903931152075529, "epoch": 0.27622638294246704, "frac_reward_zero_std": 0.625, "grad_norm": 0.4146186113357544, "learning_rate": 1e-06, "loss": 0.0003, "num_tokens": 647324555.0, "reward": 0.4765625, "reward_std": 0.11840169876813889, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1621, "tools/generated_tokens": 3397.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1012.90625, "completions/mean_terminated_length": 1004.7559204101562, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.09020825754851103, "epoch": 0.27639678786716937, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4705333709716797, "learning_rate": 1e-06, "loss": -0.0206, "num_tokens": 647639939.0, "reward": 0.58203125, "reward_std": 0.20200955867767334, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1622, "tools/generated_tokens": 2532.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1016.91796875, "completions/mean_terminated_length": 1000.5516357421875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.08934870082885027, "epoch": 0.2765671927918717, "frac_reward_zero_std": 0.625, "grad_norm": 0.5121764540672302, "learning_rate": 1e-06, "loss": 0.0103, "num_tokens": 647970782.0, "reward": 0.5390625, "reward_std": 0.1528470814228058, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1623, "tools/generated_tokens": 3488.921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1151.80078125, "completions/mean_terminated_length": 1137.575439453125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.09406989719718695, "epoch": 0.276737597716574, "frac_reward_zero_std": 0.6875, "grad_norm": 0.38163524866104126, "learning_rate": 1e-06, "loss": -0.0113, "num_tokens": 648336555.0, "reward": 0.4140625, "reward_std": 0.13663378357887268, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1624, "tools/generated_tokens": 3047.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1008.8515625, "completions/mean_terminated_length": 988.1514282226562, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.08761391788721085, "epoch": 0.27690800264127635, "frac_reward_zero_std": 0.5, "grad_norm": 0.5968549847602844, "learning_rate": 1e-06, "loss": 0.0369, "num_tokens": 648683125.0, "reward": 0.62109375, "reward_std": 0.19641819596290588, "rewards/simpleverify_reward/mean": 0.62109375, "rewards/simpleverify_reward/std": 0.4860650300979614, "step": 1625, "tools/generated_tokens": 2760.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1096.01953125, "completions/mean_terminated_length": 1080.9088134765625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.09870099555701017, "epoch": 0.2770784075659787, "frac_reward_zero_std": 0.625, "grad_norm": 0.38089367747306824, "learning_rate": 1e-06, "loss": -0.0195, "num_tokens": 649039226.0, "reward": 0.57421875, "reward_std": 0.12709103524684906, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1626, "tools/generated_tokens": 3168.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 898.6875, "completions/mean_terminated_length": 894.180419921875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.0938123669475317, "epoch": 0.277248812490681, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5537983179092407, "learning_rate": 1e-06, "loss": -0.021, "num_tokens": 649342154.0, "reward": 0.58984375, "reward_std": 0.1589793860912323, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1627, "tools/generated_tokens": 3138.69140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1158.15234375, "completions/mean_terminated_length": 1147.600830078125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.08369949972257018, "epoch": 0.2774192174153833, "frac_reward_zero_std": 0.6875, "grad_norm": 0.34804999828338623, "learning_rate": 1e-06, "loss": 0.0163, "num_tokens": 649694097.0, "reward": 0.6015625, "reward_std": 0.1281953752040863, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1628, "tools/generated_tokens": 2662.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1069.7109375, "completions/mean_terminated_length": 1065.8746337890625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.10007403744384646, "epoch": 0.2775896223400856, "frac_reward_zero_std": 0.5, "grad_norm": 0.45562899112701416, "learning_rate": 1e-06, "loss": -0.0257, "num_tokens": 650029447.0, "reward": 0.71484375, "reward_std": 0.2180052399635315, "rewards/simpleverify_reward/mean": 0.71484375, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 1629, "tools/generated_tokens": 2637.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1051.9921875, "completions/mean_terminated_length": 1051.9921875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.09410482924431562, "epoch": 0.27776002726478793, "frac_reward_zero_std": 0.625, "grad_norm": 0.5375993847846985, "learning_rate": 1e-06, "loss": -0.0403, "num_tokens": 650361493.0, "reward": 0.453125, "reward_std": 0.1650887131690979, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1630, "tools/generated_tokens": 2723.99609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.81640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1012.9765625, "completions/mean_terminated_length": 996.5476684570312, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.08191181952133775, "epoch": 0.27793043218949026, "frac_reward_zero_std": 0.3125, "grad_norm": 0.46228188276290894, "learning_rate": 1e-06, "loss": -0.0332, "num_tokens": 650690783.0, "reward": 0.63671875, "reward_std": 0.24397152662277222, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1631, "tools/generated_tokens": 3172.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1099.13671875, "completions/mean_terminated_length": 1072.4617919921875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.09068663278594613, "epoch": 0.2781008371141926, "frac_reward_zero_std": 0.6875, "grad_norm": 0.30640119314193726, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 651053746.0, "reward": 0.3046875, "reward_std": 0.1232442855834961, "rewards/simpleverify_reward/mean": 0.3046875, "rewards/simpleverify_reward/std": 0.4611765742301941, "step": 1632, "tools/generated_tokens": 4075.265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1075.49609375, "completions/mean_terminated_length": 1063.9644775390625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.085804826579988, "epoch": 0.2782712420388949, "frac_reward_zero_std": 0.625, "grad_norm": 0.3525437116622925, "learning_rate": 1e-06, "loss": -0.0082, "num_tokens": 651395857.0, "reward": 0.57421875, "reward_std": 0.16813471913337708, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1633, "tools/generated_tokens": 3059.49609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1006.60546875, "completions/mean_terminated_length": 981.612060546875, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.10488267568871379, "epoch": 0.27844164696359724, "frac_reward_zero_std": 0.125, "grad_norm": 0.6728838086128235, "learning_rate": 1e-06, "loss": -0.0429, "num_tokens": 651732156.0, "reward": 0.68359375, "reward_std": 0.33850714564323425, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1634, "tools/generated_tokens": 3766.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 983.98828125, "completions/mean_terminated_length": 979.8157348632812, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.09469101438298821, "epoch": 0.27861205188829957, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3643794357776642, "learning_rate": 1e-06, "loss": -0.0006, "num_tokens": 652050745.0, "reward": 0.48828125, "reward_std": 0.128742977976799, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1635, "tools/generated_tokens": 3031.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1079.0390625, "completions/mean_terminated_length": 1031.3851318359375, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.08922412153333426, "epoch": 0.2787824568130019, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5217407941818237, "learning_rate": 1e-06, "loss": -0.0443, "num_tokens": 652402003.0, "reward": 0.5078125, "reward_std": 0.2936214506626129, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1636, "tools/generated_tokens": 3663.04296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1181.2265625, "completions/mean_terminated_length": 1153.26611328125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.09095259942114353, "epoch": 0.2789528617377042, "frac_reward_zero_std": 0.375, "grad_norm": 0.4016534090042114, "learning_rate": 1e-06, "loss": -0.0325, "num_tokens": 652767501.0, "reward": 0.625, "reward_std": 0.22360859811306, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1637, "tools/generated_tokens": 3077.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 966.48828125, "completions/mean_terminated_length": 962.2471313476562, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.10496659949421883, "epoch": 0.27912326666240656, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5560580492019653, "learning_rate": 1e-06, "loss": -0.0267, "num_tokens": 653088362.0, "reward": 0.57421875, "reward_std": 0.1971760094165802, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1638, "tools/generated_tokens": 3214.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 944.5078125, "completions/mean_terminated_length": 940.180419921875, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.10365258762612939, "epoch": 0.2792936715871089, "frac_reward_zero_std": 0.375, "grad_norm": 0.6258965730667114, "learning_rate": 1e-06, "loss": -0.0441, "num_tokens": 653406396.0, "reward": 0.5, "reward_std": 0.23260819911956787, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1639, "tools/generated_tokens": 3520.515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1153.734375, "completions/mean_terminated_length": 1139.543701171875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.10484739486128092, "epoch": 0.2794640765118112, "frac_reward_zero_std": 0.5, "grad_norm": 0.4401443600654602, "learning_rate": 1e-06, "loss": 0.0332, "num_tokens": 653775448.0, "reward": 0.578125, "reward_std": 0.17971593141555786, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1640, "tools/generated_tokens": 3705.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1085.71875, "completions/mean_terminated_length": 1054.6773681640625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.10529569210484624, "epoch": 0.27963448143651354, "frac_reward_zero_std": 0.8125, "grad_norm": 0.3253202736377716, "learning_rate": 1e-06, "loss": -0.0259, "num_tokens": 654126544.0, "reward": 0.38671875, "reward_std": 0.046875, "rewards/simpleverify_reward/mean": 0.38671875, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1641, "tools/generated_tokens": 3373.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1073.95703125, "completions/mean_terminated_length": 1070.1373291015625, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.10323024401441216, "epoch": 0.27980488636121587, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3857119083404541, "learning_rate": 1e-06, "loss": 0.0391, "num_tokens": 654468309.0, "reward": 0.56640625, "reward_std": 0.17341843247413635, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1642, "tools/generated_tokens": 3009.953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1088.734375, "completions/mean_terminated_length": 1061.7669677734375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.10303151933476329, "epoch": 0.27997529128591814, "frac_reward_zero_std": 0.375, "grad_norm": 0.373665452003479, "learning_rate": 1e-06, "loss": 0.0072, "num_tokens": 654824273.0, "reward": 0.3984375, "reward_std": 0.23516272008419037, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1643, "tools/generated_tokens": 3656.7421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1085.546875, "completions/mean_terminated_length": 1021.3833618164062, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.10210465965792537, "epoch": 0.28014569621062047, "frac_reward_zero_std": 0.1875, "grad_norm": 0.4981166422367096, "learning_rate": 1e-06, "loss": -0.0356, "num_tokens": 655187693.0, "reward": 0.47265625, "reward_std": 0.3232673406600952, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1644, "tools/generated_tokens": 4125.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1254.328125, "completions/mean_terminated_length": 1201.416748046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.10195873165503144, "epoch": 0.2803161011353228, "frac_reward_zero_std": 0.5625, "grad_norm": 0.475806325674057, "learning_rate": 1e-06, "loss": 0.0358, "num_tokens": 655575041.0, "reward": 0.5859375, "reward_std": 0.174540713429451, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1645, "tools/generated_tokens": 3478.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1169.48828125, "completions/mean_terminated_length": 1166.043212890625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.1068618563003838, "epoch": 0.2804865060600251, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4178864657878876, "learning_rate": 1e-06, "loss": 0.0302, "num_tokens": 655952478.0, "reward": 0.46875, "reward_std": 0.23875631392002106, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1646, "tools/generated_tokens": 3409.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1034.63671875, "completions/mean_terminated_length": 1006.1485595703125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.12260621739551425, "epoch": 0.28065691098472745, "frac_reward_zero_std": 0.3125, "grad_norm": 0.8791807293891907, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 656289217.0, "reward": 0.51953125, "reward_std": 0.23381631076335907, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1647, "tools/generated_tokens": 3714.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1031.49609375, "completions/mean_terminated_length": 1015.3611450195312, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.11728832172229886, "epoch": 0.2808273159094298, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4330569803714752, "learning_rate": 1e-06, "loss": 0.0339, "num_tokens": 656630384.0, "reward": 0.484375, "reward_std": 0.21181908249855042, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1648, "tools/generated_tokens": 3575.50390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1084.84375, "completions/mean_terminated_length": 1077.2598876953125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.11593197379261255, "epoch": 0.2809977208341321, "frac_reward_zero_std": 0.5, "grad_norm": 0.3523751199245453, "learning_rate": 1e-06, "loss": -0.0119, "num_tokens": 656983992.0, "reward": 0.48828125, "reward_std": 0.1759347915649414, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1649, "tools/generated_tokens": 3740.8515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1182.51953125, "completions/mean_terminated_length": 1154.600830078125, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.09640831965953112, "epoch": 0.28116812575883443, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4158303141593933, "learning_rate": 1e-06, "loss": -0.0145, "num_tokens": 657341837.0, "reward": 0.59765625, "reward_std": 0.1581149846315384, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1650, "tools/generated_tokens": 2806.51953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.79296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 931.73046875, "completions/mean_terminated_length": 909.4940795898438, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.11387848854064941, "epoch": 0.28133853068353676, "frac_reward_zero_std": 0.375, "grad_norm": 0.40488001704216003, "learning_rate": 1e-06, "loss": 0.0055, "num_tokens": 657650008.0, "reward": 0.69140625, "reward_std": 0.22797390818595886, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1651, "tools/generated_tokens": 2747.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.88671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1068.71484375, "completions/mean_terminated_length": 1061.00390625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.0994862006045878, "epoch": 0.2815089356082391, "frac_reward_zero_std": 0.625, "grad_norm": 0.40518856048583984, "learning_rate": 1e-06, "loss": -0.0298, "num_tokens": 658001519.0, "reward": 0.79296875, "reward_std": 0.11091843992471695, "rewards/simpleverify_reward/mean": 0.79296875, "rewards/simpleverify_reward/std": 0.40597182512283325, "step": 1652, "tools/generated_tokens": 3044.72265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1048.4453125, "completions/mean_terminated_length": 1016.2015991210938, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.08903263928368688, "epoch": 0.2816793405329414, "frac_reward_zero_std": 0.75, "grad_norm": 0.30986666679382324, "learning_rate": 1e-06, "loss": 0.0089, "num_tokens": 658327457.0, "reward": 0.55078125, "reward_std": 0.11058580875396729, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1653, "tools/generated_tokens": 2800.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1061.8671875, "completions/mean_terminated_length": 1058.0001220703125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.11921488214284182, "epoch": 0.28184974545764374, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4146956205368042, "learning_rate": 1e-06, "loss": -0.0198, "num_tokens": 658676687.0, "reward": 0.4375, "reward_std": 0.1588183045387268, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1654, "tools/generated_tokens": 3597.87109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2007.0, "completions/mean_length": 1120.1015625, "completions/mean_terminated_length": 1082.382080078125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.10010175919160247, "epoch": 0.28202015038234607, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3490271270275116, "learning_rate": 1e-06, "loss": 0.0202, "num_tokens": 659048281.0, "reward": 0.5859375, "reward_std": 0.14789125323295593, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1655, "tools/generated_tokens": 3624.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1124.3203125, "completions/mean_terminated_length": 1117.0472412109375, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.10063887387514114, "epoch": 0.2821905553070484, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3113102614879608, "learning_rate": 1e-06, "loss": -0.0138, "num_tokens": 659401163.0, "reward": 0.48046875, "reward_std": 0.18816794455051422, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1656, "tools/generated_tokens": 2900.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1030.14453125, "completions/mean_terminated_length": 997.3104248046875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.1053746659308672, "epoch": 0.2823609602317507, "frac_reward_zero_std": 0.375, "grad_norm": 0.48907846212387085, "learning_rate": 1e-06, "loss": 0.0129, "num_tokens": 659742592.0, "reward": 0.515625, "reward_std": 0.23745271563529968, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1657, "tools/generated_tokens": 3238.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1021.65234375, "completions/mean_terminated_length": 979.9308471679688, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.08726793946698308, "epoch": 0.282531365156453, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4810398519039154, "learning_rate": 1e-06, "loss": -0.0066, "num_tokens": 660088071.0, "reward": 0.625, "reward_std": 0.2403016984462738, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1658, "tools/generated_tokens": 3741.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1148.67578125, "completions/mean_terminated_length": 1141.594482421875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0866245049983263, "epoch": 0.2827017700811553, "frac_reward_zero_std": 0.375, "grad_norm": 0.5404751896858215, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 660447588.0, "reward": 0.59765625, "reward_std": 0.22887900471687317, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1659, "tools/generated_tokens": 2548.6640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1192.9140625, "completions/mean_terminated_length": 1158.1544189453125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.09302658215165138, "epoch": 0.28287217500585765, "frac_reward_zero_std": 0.5, "grad_norm": 0.38092970848083496, "learning_rate": 1e-06, "loss": 0.0022, "num_tokens": 660821694.0, "reward": 0.5390625, "reward_std": 0.19885504245758057, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1660, "tools/generated_tokens": 3464.91796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1081.6484375, "completions/mean_terminated_length": 1081.6484375, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.08700334886088967, "epoch": 0.28304257993056, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3749883472919464, "learning_rate": 1e-06, "loss": -0.0154, "num_tokens": 661157972.0, "reward": 0.51953125, "reward_std": 0.23559054732322693, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1661, "tools/generated_tokens": 2641.6484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.76171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1109.0625, "completions/mean_terminated_length": 1078.774169921875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.08288721228018403, "epoch": 0.2832129848552623, "frac_reward_zero_std": 0.625, "grad_norm": 0.3786908984184265, "learning_rate": 1e-06, "loss": 0.0069, "num_tokens": 661521716.0, "reward": 0.75, "reward_std": 0.16241663694381714, "rewards/simpleverify_reward/mean": 0.75, "rewards/simpleverify_reward/std": 0.4338609278202057, "step": 1662, "tools/generated_tokens": 3669.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1112.3125, "completions/mean_terminated_length": 1074.2803955078125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.07103221258148551, "epoch": 0.28338338977996463, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3213314712047577, "learning_rate": 1e-06, "loss": 0.0252, "num_tokens": 661881876.0, "reward": 0.4921875, "reward_std": 0.17396603524684906, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1663, "tools/generated_tokens": 3576.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1130.28125, "completions/mean_terminated_length": 1119.3992919921875, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.07078636810183525, "epoch": 0.28355379470466696, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3718040883541107, "learning_rate": 1e-06, "loss": 0.0029, "num_tokens": 662242476.0, "reward": 0.48828125, "reward_std": 0.15943148732185364, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1664, "tools/generated_tokens": 2930.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1055.83203125, "completions/mean_terminated_length": 1044.0672607421875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.07396357948891819, "epoch": 0.2837241996293693, "frac_reward_zero_std": 0.4375, "grad_norm": 0.39329102635383606, "learning_rate": 1e-06, "loss": 0.0328, "num_tokens": 662579169.0, "reward": 0.4296875, "reward_std": 0.21933844685554504, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1665, "tools/generated_tokens": 2927.83984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1160.50390625, "completions/mean_terminated_length": 1113.024658203125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.07150605623610318, "epoch": 0.2838946045540716, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4315892457962036, "learning_rate": 1e-06, "loss": 0.0053, "num_tokens": 662940914.0, "reward": 0.51953125, "reward_std": 0.16518136858940125, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1666, "tools/generated_tokens": 3080.51171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1094.75, "completions/mean_terminated_length": 1051.950927734375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.06752468412742019, "epoch": 0.28406500947877394, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3731021285057068, "learning_rate": 1e-06, "loss": 0.0149, "num_tokens": 663279506.0, "reward": 0.66015625, "reward_std": 0.17296825349330902, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1667, "tools/generated_tokens": 3102.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.98046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1034.31640625, "completions/mean_terminated_length": 1014.12353515625, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.06526755052618682, "epoch": 0.28423541440347627, "frac_reward_zero_std": 0.1875, "grad_norm": 0.5776063799858093, "learning_rate": 1e-06, "loss": -0.0116, "num_tokens": 663626755.0, "reward": 0.41015625, "reward_std": 0.30510586500167847, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1668, "tools/generated_tokens": 3354.3125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2016.0, "completions/mean_length": 1070.3203125, "completions/mean_terminated_length": 1046.8560791015625, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.06758396839722991, "epoch": 0.2844058193281786, "frac_reward_zero_std": 0.625, "grad_norm": 0.3736812472343445, "learning_rate": 1e-06, "loss": 0.0052, "num_tokens": 663967797.0, "reward": 0.6875, "reward_std": 0.16613000631332397, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1669, "tools/generated_tokens": 3102.31640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1087.84375, "completions/mean_terminated_length": 1076.45849609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.06929038930684328, "epoch": 0.2845762242528809, "frac_reward_zero_std": 0.5, "grad_norm": 0.43488645553588867, "learning_rate": 1e-06, "loss": -0.0001, "num_tokens": 664319853.0, "reward": 0.7265625, "reward_std": 0.21189872920513153, "rewards/simpleverify_reward/mean": 0.7265625, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 1670, "tools/generated_tokens": 2991.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1176.0859375, "completions/mean_terminated_length": 1121.8258056640625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.07867563189938664, "epoch": 0.28474662917758325, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3505718410015106, "learning_rate": 1e-06, "loss": -0.0023, "num_tokens": 664698563.0, "reward": 0.4140625, "reward_std": 0.17021197080612183, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1671, "tools/generated_tokens": 4320.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1156.94140625, "completions/mean_terminated_length": 1085.5062255859375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07630252419039607, "epoch": 0.2849170341022856, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3492695391178131, "learning_rate": 1e-06, "loss": -0.0156, "num_tokens": 665063236.0, "reward": 0.44921875, "reward_std": 0.14006631076335907, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1672, "tools/generated_tokens": 3748.94140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1032.9765625, "completions/mean_terminated_length": 1016.8651123046875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.07324122404679656, "epoch": 0.28508743902698785, "frac_reward_zero_std": 0.4375, "grad_norm": 0.9570232033729553, "learning_rate": 1e-06, "loss": -0.0049, "num_tokens": 665384062.0, "reward": 0.70703125, "reward_std": 0.24352134764194489, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 1673, "tools/generated_tokens": 2640.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.78515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1031.43359375, "completions/mean_terminated_length": 1031.43359375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.0778612308204174, "epoch": 0.2852578439516902, "frac_reward_zero_std": 0.6875, "grad_norm": 0.4046790599822998, "learning_rate": 1e-06, "loss": 0.0165, "num_tokens": 665707629.0, "reward": 0.69140625, "reward_std": 0.1318160742521286, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1674, "tools/generated_tokens": 2671.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1108.97265625, "completions/mean_terminated_length": 1042.188232421875, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.07713081128895283, "epoch": 0.2854282488763925, "frac_reward_zero_std": 0.5, "grad_norm": 0.388851135969162, "learning_rate": 1e-06, "loss": 0.02, "num_tokens": 666067750.0, "reward": 0.46875, "reward_std": 0.20443321764469147, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1675, "tools/generated_tokens": 4148.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1093.3984375, "completions/mean_terminated_length": 1074.3824462890625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.07440097071230412, "epoch": 0.28559865380109484, "frac_reward_zero_std": 0.5625, "grad_norm": 0.46859118342399597, "learning_rate": 1e-06, "loss": -0.0019, "num_tokens": 666405820.0, "reward": 0.609375, "reward_std": 0.13423693180084229, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1676, "tools/generated_tokens": 2565.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1094.83984375, "completions/mean_terminated_length": 1071.964111328125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.07315214560367167, "epoch": 0.28576905872579716, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5854039788246155, "learning_rate": 1e-06, "loss": -0.0091, "num_tokens": 666759507.0, "reward": 0.6328125, "reward_std": 0.17686696350574493, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1677, "tools/generated_tokens": 3406.8359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1076.88671875, "completions/mean_terminated_length": 1069.2401123046875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07078006980009377, "epoch": 0.2859394636504995, "frac_reward_zero_std": 0.5, "grad_norm": 0.47656646370887756, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 667105686.0, "reward": 0.53515625, "reward_std": 0.2057664394378662, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1678, "tools/generated_tokens": 2988.89453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.93359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2045.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1061.453125, "completions/mean_terminated_length": 1061.453125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.06865523639135063, "epoch": 0.2861098685752018, "frac_reward_zero_std": 0.5625, "grad_norm": 0.41567087173461914, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 667433626.0, "reward": 0.55859375, "reward_std": 0.17252904176712036, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1679, "tools/generated_tokens": 2429.45703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1141.3671875, "completions/mean_terminated_length": 1123.3067626953125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.07780265994369984, "epoch": 0.28628027349990415, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4451778531074524, "learning_rate": 1e-06, "loss": 0.0415, "num_tokens": 667786328.0, "reward": 0.6640625, "reward_std": 0.1706668734550476, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 1680, "tools/generated_tokens": 2893.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.85546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 985.02734375, "completions/mean_terminated_length": 980.85888671875, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.07236904790624976, "epoch": 0.2864506784246065, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4616529643535614, "learning_rate": 1e-06, "loss": -0.0037, "num_tokens": 668110703.0, "reward": 0.73828125, "reward_std": 0.19210736453533173, "rewards/simpleverify_reward/mean": 0.73828125, "rewards/simpleverify_reward/std": 0.4404313564300537, "step": 1681, "tools/generated_tokens": 3233.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1182.1328125, "completions/mean_terminated_length": 1154.2015380859375, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.07440485386177897, "epoch": 0.2866210833493088, "frac_reward_zero_std": 0.375, "grad_norm": 0.5135686993598938, "learning_rate": 1e-06, "loss": -0.0195, "num_tokens": 668488833.0, "reward": 0.7109375, "reward_std": 0.21575656533241272, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1682, "tools/generated_tokens": 3734.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1050.3671875, "completions/mean_terminated_length": 1038.53759765625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.07658576592803001, "epoch": 0.28679148827401113, "frac_reward_zero_std": 0.3125, "grad_norm": 0.519871711730957, "learning_rate": 1e-06, "loss": -0.0334, "num_tokens": 668827359.0, "reward": 0.55078125, "reward_std": 0.2540486752986908, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1683, "tools/generated_tokens": 3514.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1095.1484375, "completions/mean_terminated_length": 1083.849853515625, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.07869702484458685, "epoch": 0.28696189319871346, "frac_reward_zero_std": 0.4375, "grad_norm": 0.47786691784858704, "learning_rate": 1e-06, "loss": -0.0177, "num_tokens": 669183333.0, "reward": 0.49609375, "reward_std": 0.17120973765850067, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1684, "tools/generated_tokens": 3767.16796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1097.5546875, "completions/mean_terminated_length": 1062.923095703125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08632402727380395, "epoch": 0.2871322981234158, "frac_reward_zero_std": 0.625, "grad_norm": 0.4157116115093231, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 669543395.0, "reward": 0.33984375, "reward_std": 0.14604227244853973, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1685, "tools/generated_tokens": 3577.55859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2044.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1044.9296875, "completions/mean_terminated_length": 1044.9296875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.07719290442764759, "epoch": 0.2873027030481181, "frac_reward_zero_std": 0.5, "grad_norm": 0.4584277868270874, "learning_rate": 1e-06, "loss": 0.0086, "num_tokens": 669874097.0, "reward": 0.5546875, "reward_std": 0.2210540771484375, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1686, "tools/generated_tokens": 2940.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1111.390625, "completions/mean_terminated_length": 1096.52392578125, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.07375420350581408, "epoch": 0.28747310797282044, "frac_reward_zero_std": 0.6875, "grad_norm": 0.4551894962787628, "learning_rate": 1e-06, "loss": 0.018, "num_tokens": 670212245.0, "reward": 0.546875, "reward_std": 0.11904004216194153, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1687, "tools/generated_tokens": 2799.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.82421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1101.8828125, "completions/mean_terminated_length": 1090.6640625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.08003143081441522, "epoch": 0.2876435128975227, "frac_reward_zero_std": 0.6875, "grad_norm": 0.417856901884079, "learning_rate": 1e-06, "loss": -0.0079, "num_tokens": 670554967.0, "reward": 0.625, "reward_std": 0.12673160433769226, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1688, "tools/generated_tokens": 3237.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1041.08203125, "completions/mean_terminated_length": 1041.08203125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07794794300571084, "epoch": 0.28781391782222504, "frac_reward_zero_std": 0.25, "grad_norm": 0.5890607833862305, "learning_rate": 1e-06, "loss": 0.0288, "num_tokens": 670904060.0, "reward": 0.5859375, "reward_std": 0.27924326062202454, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1689, "tools/generated_tokens": 3657.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1209.83203125, "completions/mean_terminated_length": 1127.0943603515625, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0808892878703773, "epoch": 0.28798432274692737, "frac_reward_zero_std": 0.375, "grad_norm": 0.44910311698913574, "learning_rate": 1e-06, "loss": 0.0055, "num_tokens": 671288817.0, "reward": 0.24609375, "reward_std": 0.23742592334747314, "rewards/simpleverify_reward/mean": 0.24609375, "rewards/simpleverify_reward/std": 0.43157756328582764, "step": 1690, "tools/generated_tokens": 4705.83984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.70703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2035.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1006.484375, "completions/mean_terminated_length": 1006.484375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.08054341049864888, "epoch": 0.2881547276716297, "frac_reward_zero_std": 0.625, "grad_norm": 0.5780123472213745, "learning_rate": 1e-06, "loss": -0.0104, "num_tokens": 671599965.0, "reward": 0.58203125, "reward_std": 0.13290652632713318, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1691, "tools/generated_tokens": 2350.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.65625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1050.6953125, "completions/mean_terminated_length": 1046.784423828125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.08420492149889469, "epoch": 0.288325132596332, "frac_reward_zero_std": 0.6875, "grad_norm": 0.30409345030784607, "learning_rate": 1e-06, "loss": 0.0306, "num_tokens": 671939151.0, "reward": 0.3828125, "reward_std": 0.1399868130683899, "rewards/simpleverify_reward/mean": 0.3828125, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1692, "tools/generated_tokens": 3514.7265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1154.36328125, "completions/mean_terminated_length": 1132.916015625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.0903378939256072, "epoch": 0.28849553752103435, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5356123447418213, "learning_rate": 1e-06, "loss": 0.0185, "num_tokens": 672306908.0, "reward": 0.52734375, "reward_std": 0.20735841989517212, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1693, "tools/generated_tokens": 3194.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2035.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1037.13671875, "completions/mean_terminated_length": 1037.13671875, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.08119894750416279, "epoch": 0.2886659424457367, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5017520189285278, "learning_rate": 1e-06, "loss": 0.03, "num_tokens": 672635183.0, "reward": 0.68359375, "reward_std": 0.27668559551239014, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1694, "tools/generated_tokens": 2645.13671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.78515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1137.33984375, "completions/mean_terminated_length": 1133.7686767578125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.07596439821645617, "epoch": 0.288836347370439, "frac_reward_zero_std": 0.625, "grad_norm": 0.3324894607067108, "learning_rate": 1e-06, "loss": -0.0073, "num_tokens": 672993734.0, "reward": 0.5859375, "reward_std": 0.14523237943649292, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1695, "tools/generated_tokens": 2953.34375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.88671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1062.14453125, "completions/mean_terminated_length": 1042.5059814453125, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.07371445535682142, "epoch": 0.28900675229514133, "frac_reward_zero_std": 0.5625, "grad_norm": 0.41627153754234314, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 673320795.0, "reward": 0.69140625, "reward_std": 0.1638239026069641, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1696, "tools/generated_tokens": 2774.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2044.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1032.3046875, "completions/mean_terminated_length": 1032.3046875, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.0861100135371089, "epoch": 0.28917715721984366, "frac_reward_zero_std": 0.75, "grad_norm": 0.36681443452835083, "learning_rate": 1e-06, "loss": -0.0186, "num_tokens": 673643673.0, "reward": 0.66796875, "reward_std": 0.11109209060668945, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1697, "tools/generated_tokens": 2664.3125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1027.05078125, "completions/mean_terminated_length": 1023.047119140625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08342539612203836, "epoch": 0.289347562144546, "frac_reward_zero_std": 0.625, "grad_norm": 0.4858676791191101, "learning_rate": 1e-06, "loss": 0.0056, "num_tokens": 673967814.0, "reward": 0.66796875, "reward_std": 0.14029237627983093, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1698, "tools/generated_tokens": 3051.06640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.98828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 983.25, "completions/mean_terminated_length": 983.25, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.0876630898565054, "epoch": 0.2895179670692483, "frac_reward_zero_std": 0.375, "grad_norm": 0.613278329372406, "learning_rate": 1e-06, "loss": 0.0014, "num_tokens": 674295286.0, "reward": 0.66015625, "reward_std": 0.22932013869285583, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1699, "tools/generated_tokens": 2943.2421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1075.09375, "completions/mean_terminated_length": 1047.7509765625, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.0845459895208478, "epoch": 0.28968837199395064, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3851766288280487, "learning_rate": 1e-06, "loss": -0.0246, "num_tokens": 674649006.0, "reward": 0.46484375, "reward_std": 0.19290617108345032, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1700, "tools/generated_tokens": 3691.1015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1254.19921875, "completions/mean_terminated_length": 1218.55908203125, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.0739209046587348, "epoch": 0.28985877691865297, "frac_reward_zero_std": 0.5, "grad_norm": 0.37351107597351074, "learning_rate": 1e-06, "loss": 0.0063, "num_tokens": 675029233.0, "reward": 0.47265625, "reward_std": 0.16078896820545197, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1701, "tools/generated_tokens": 3342.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1054.796875, "completions/mean_terminated_length": 1043.0238037109375, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.07070727366954088, "epoch": 0.2900291818433553, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5558307766914368, "learning_rate": 1e-06, "loss": -0.0023, "num_tokens": 675368349.0, "reward": 0.640625, "reward_std": 0.255313515663147, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1702, "tools/generated_tokens": 3286.83203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.08984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 933.98046875, "completions/mean_terminated_length": 933.98046875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.08964383602142334, "epoch": 0.29019958676805757, "frac_reward_zero_std": 0.5625, "grad_norm": 0.42618507146835327, "learning_rate": 1e-06, "loss": -0.0118, "num_tokens": 675693544.0, "reward": 0.41796875, "reward_std": 0.15976691246032715, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1703, "tools/generated_tokens": 3781.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1023.11328125, "completions/mean_terminated_length": 1019.0941772460938, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.07676013419404626, "epoch": 0.2903699916927599, "frac_reward_zero_std": 0.5625, "grad_norm": 0.36840879917144775, "learning_rate": 1e-06, "loss": 0.0187, "num_tokens": 676027381.0, "reward": 0.4296875, "reward_std": 0.16394630074501038, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1704, "tools/generated_tokens": 3759.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2045.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1024.21484375, "completions/mean_terminated_length": 1024.21484375, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.07335200719535351, "epoch": 0.2905403966174622, "frac_reward_zero_std": 0.625, "grad_norm": 0.40544983744621277, "learning_rate": 1e-06, "loss": -0.0449, "num_tokens": 676357100.0, "reward": 0.72265625, "reward_std": 0.1254390925168991, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1705, "tools/generated_tokens": 3000.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1004.54296875, "completions/mean_terminated_length": 992.1699829101562, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.07652433076873422, "epoch": 0.29071080154216455, "frac_reward_zero_std": 0.625, "grad_norm": 0.4287164509296417, "learning_rate": 1e-06, "loss": -0.0005, "num_tokens": 676694503.0, "reward": 0.46875, "reward_std": 0.14407353103160858, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1706, "tools/generated_tokens": 4052.55078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1027.98046875, "completions/mean_terminated_length": 1027.98046875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.09149813745170832, "epoch": 0.2908812064668669, "frac_reward_zero_std": 0.5, "grad_norm": 0.5194151401519775, "learning_rate": 1e-06, "loss": -0.0238, "num_tokens": 677038962.0, "reward": 0.4375, "reward_std": 0.16974535584449768, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1707, "tools/generated_tokens": 3683.98046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2029.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1022.29296875, "completions/mean_terminated_length": 1022.29296875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.08804600546136498, "epoch": 0.2910516113915692, "frac_reward_zero_std": 0.5, "grad_norm": 0.4183364808559418, "learning_rate": 1e-06, "loss": 0.0249, "num_tokens": 677385901.0, "reward": 0.52734375, "reward_std": 0.19045542180538177, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1708, "tools/generated_tokens": 3270.296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1087.70703125, "completions/mean_terminated_length": 1068.5777587890625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.08538332208991051, "epoch": 0.29122201631627154, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4747770428657532, "learning_rate": 1e-06, "loss": 0.0128, "num_tokens": 677734306.0, "reward": 0.46484375, "reward_std": 0.25074952840805054, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1709, "tools/generated_tokens": 3463.7109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.16015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1133.30859375, "completions/mean_terminated_length": 1111.3560791015625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07822273066267371, "epoch": 0.29139242124097386, "frac_reward_zero_std": 0.375, "grad_norm": 0.5086626410484314, "learning_rate": 1e-06, "loss": -0.0076, "num_tokens": 678096881.0, "reward": 0.5390625, "reward_std": 0.2718871235847473, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1710, "tools/generated_tokens": 4037.31640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1063.19140625, "completions/mean_terminated_length": 1027.3077392578125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08961726538836956, "epoch": 0.2915628261656762, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4602007269859314, "learning_rate": 1e-06, "loss": -0.0112, "num_tokens": 678436514.0, "reward": 0.671875, "reward_std": 0.1849614828824997, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1711, "tools/generated_tokens": 3183.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.03515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1141.90625, "completions/mean_terminated_length": 1138.35302734375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.08629632648080587, "epoch": 0.2917332310903785, "frac_reward_zero_std": 0.3125, "grad_norm": 0.38825368881225586, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 678796202.0, "reward": 0.7109375, "reward_std": 0.2651072144508362, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1712, "tools/generated_tokens": 3421.91015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.11328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1179.64453125, "completions/mean_terminated_length": 1151.6370849609375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.07957607274875045, "epoch": 0.29190363601508085, "frac_reward_zero_std": 0.125, "grad_norm": 0.5259243249893188, "learning_rate": 1e-06, "loss": 0.0132, "num_tokens": 679183471.0, "reward": 0.53515625, "reward_std": 0.33970892429351807, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1713, "tools/generated_tokens": 3907.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 972.07421875, "completions/mean_terminated_length": 941.8272705078125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.08360702497884631, "epoch": 0.2920740409397832, "frac_reward_zero_std": 0.5625, "grad_norm": 0.40394261479377747, "learning_rate": 1e-06, "loss": 0.0246, "num_tokens": 679514482.0, "reward": 0.73046875, "reward_std": 0.16915957629680634, "rewards/simpleverify_reward/mean": 0.73046875, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 1714, "tools/generated_tokens": 3564.078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2005.0, "completions/mean_length": 1046.28515625, "completions/mean_terminated_length": 1042.35693359375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.07854842068627477, "epoch": 0.2922444458644855, "frac_reward_zero_std": 0.5625, "grad_norm": 0.38907700777053833, "learning_rate": 1e-06, "loss": 0.0021, "num_tokens": 679847995.0, "reward": 0.55078125, "reward_std": 0.17406152188777924, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1715, "tools/generated_tokens": 2782.29296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.84765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2017.0, "completions/mean_length": 1097.0859375, "completions/mean_terminated_length": 1093.35693359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.09003267157822847, "epoch": 0.29241485078918783, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4848196804523468, "learning_rate": 1e-06, "loss": 0.0183, "num_tokens": 680197217.0, "reward": 0.59765625, "reward_std": 0.26143452525138855, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1716, "tools/generated_tokens": 2985.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1136.1015625, "completions/mean_terminated_length": 1125.28857421875, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.07964895945042372, "epoch": 0.29258525571389016, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4489302635192871, "learning_rate": 1e-06, "loss": -0.0116, "num_tokens": 680563067.0, "reward": 0.45703125, "reward_std": 0.21818022429943085, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1717, "tools/generated_tokens": 3888.1015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.34375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1094.6875, "completions/mean_terminated_length": 1079.5556640625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07605834491550922, "epoch": 0.29275566063859243, "frac_reward_zero_std": 0.5, "grad_norm": 0.34152066707611084, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 680913403.0, "reward": 0.62890625, "reward_std": 0.20820963382720947, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1718, "tools/generated_tokens": 3054.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.95703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1134.87890625, "completions/mean_terminated_length": 1124.0513916015625, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.07997099356725812, "epoch": 0.29292606556329476, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3797798752784729, "learning_rate": 1e-06, "loss": -0.0154, "num_tokens": 681260844.0, "reward": 0.59375, "reward_std": 0.21108348667621613, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1719, "tools/generated_tokens": 3022.87890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1212.75, "completions/mean_terminated_length": 1178.7967529296875, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.08533276664093137, "epoch": 0.2930964704879971, "frac_reward_zero_std": 0.625, "grad_norm": 0.33870887756347656, "learning_rate": 1e-06, "loss": -0.0171, "num_tokens": 681651724.0, "reward": 0.5, "reward_std": 0.17835843563079834, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1720, "tools/generated_tokens": 4020.75, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1087.359375, "completions/mean_terminated_length": 1072.1112060546875, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.07737842155620456, "epoch": 0.2932668754126994, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3852839469909668, "learning_rate": 1e-06, "loss": -0.0207, "num_tokens": 682004136.0, "reward": 0.71875, "reward_std": 0.17880862951278687, "rewards/simpleverify_reward/mean": 0.71875, "rewards/simpleverify_reward/std": 0.45048993825912476, "step": 1721, "tools/generated_tokens": 3815.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1144.05078125, "completions/mean_terminated_length": 1095.6912841796875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.07036528340540826, "epoch": 0.29343728033740174, "frac_reward_zero_std": 0.6875, "grad_norm": 0.366564005613327, "learning_rate": 1e-06, "loss": 0.0178, "num_tokens": 682372613.0, "reward": 0.53515625, "reward_std": 0.12247256934642792, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1722, "tools/generated_tokens": 3816.0625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1211.62890625, "completions/mean_terminated_length": 1188.116455078125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.08202381990849972, "epoch": 0.29360768526210407, "frac_reward_zero_std": 0.5, "grad_norm": 0.40644797682762146, "learning_rate": 1e-06, "loss": 0.0052, "num_tokens": 682762822.0, "reward": 0.4296875, "reward_std": 0.19332925975322723, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1723, "tools/generated_tokens": 4403.62890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1061.58203125, "completions/mean_terminated_length": 1037.912109375, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.07391817728057504, "epoch": 0.2937780901868064, "frac_reward_zero_std": 0.5, "grad_norm": 0.4104459881782532, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 683115115.0, "reward": 0.49609375, "reward_std": 0.18914085626602173, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1724, "tools/generated_tokens": 3701.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1102.39453125, "completions/mean_terminated_length": 1035.1339111328125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.08200629102066159, "epoch": 0.2939484951115087, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6449349522590637, "learning_rate": 1e-06, "loss": 0.0493, "num_tokens": 683477136.0, "reward": 0.5859375, "reward_std": 0.22305183112621307, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1725, "tools/generated_tokens": 4134.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1140.48046875, "completions/mean_terminated_length": 1129.7193603515625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.0771291321143508, "epoch": 0.29411890003621105, "frac_reward_zero_std": 0.625, "grad_norm": 0.4248315095901489, "learning_rate": 1e-06, "loss": 0.0063, "num_tokens": 683838123.0, "reward": 0.6328125, "reward_std": 0.1528470814228058, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1726, "tools/generated_tokens": 3148.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.98046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1219.95703125, "completions/mean_terminated_length": 1164.7584228515625, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.06994751049205661, "epoch": 0.2942893049609134, "frac_reward_zero_std": 0.5625, "grad_norm": 0.36336466670036316, "learning_rate": 1e-06, "loss": 0.0157, "num_tokens": 684226592.0, "reward": 0.51171875, "reward_std": 0.17252711951732635, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1727, "tools/generated_tokens": 4123.9609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1025.7421875, "completions/mean_terminated_length": 1021.7333984375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07904585730284452, "epoch": 0.2944597098856157, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2624605894088745, "learning_rate": 1e-06, "loss": -0.021, "num_tokens": 684549710.0, "reward": 0.51953125, "reward_std": 0.11343477666378021, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1728, "tools/generated_tokens": 2689.74609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1131.89453125, "completions/mean_terminated_length": 1086.84423828125, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.0784046552143991, "epoch": 0.29463011481031803, "frac_reward_zero_std": 0.625, "grad_norm": 0.3569745123386383, "learning_rate": 1e-06, "loss": -0.0099, "num_tokens": 684910995.0, "reward": 0.45703125, "reward_std": 0.14457820355892181, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1729, "tools/generated_tokens": 3291.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1212.40234375, "completions/mean_terminated_length": 1174.8856201171875, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.06832120334729552, "epoch": 0.29480051973502036, "frac_reward_zero_std": 0.75, "grad_norm": 0.2576883137226105, "learning_rate": 1e-06, "loss": -0.0135, "num_tokens": 685290826.0, "reward": 0.28515625, "reward_std": 0.10244406759738922, "rewards/simpleverify_reward/mean": 0.28515625, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 1730, "tools/generated_tokens": 3924.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1055.3046875, "completions/mean_terminated_length": 1027.401611328125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.06822490668855608, "epoch": 0.2949709246597227, "frac_reward_zero_std": 0.375, "grad_norm": 0.5652523040771484, "learning_rate": 1e-06, "loss": -0.0072, "num_tokens": 685635672.0, "reward": 0.5078125, "reward_std": 0.2487104833126068, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1731, "tools/generated_tokens": 3655.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.26953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1125.3984375, "completions/mean_terminated_length": 1063.8917236328125, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.06617145147174597, "epoch": 0.295141329584425, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5041280388832092, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 685996830.0, "reward": 0.40234375, "reward_std": 0.19375738501548767, "rewards/simpleverify_reward/mean": 0.40234375, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1732, "tools/generated_tokens": 3765.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2032.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1122.55859375, "completions/mean_terminated_length": 1122.55859375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0621781088411808, "epoch": 0.2953117345091273, "frac_reward_zero_std": 0.9375, "grad_norm": 0.029598800465464592, "learning_rate": 1e-06, "loss": -0.0102, "num_tokens": 686337341.0, "reward": 0.68359375, "reward_std": 0.015625, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1733, "tools/generated_tokens": 2242.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1086.390625, "completions/mean_terminated_length": 1074.9881591796875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.08324898453429341, "epoch": 0.2954821394338296, "frac_reward_zero_std": 0.4375, "grad_norm": 0.43185532093048096, "learning_rate": 1e-06, "loss": -0.0238, "num_tokens": 686687825.0, "reward": 0.63671875, "reward_std": 0.1923314929008484, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1734, "tools/generated_tokens": 3398.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1018.015625, "completions/mean_terminated_length": 993.2960205078125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.07003563758917153, "epoch": 0.29565254435853194, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3594060242176056, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 687026293.0, "reward": 0.50390625, "reward_std": 0.1630660742521286, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1735, "tools/generated_tokens": 3498.01171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1076.84375, "completions/mean_terminated_length": 1045.5201416015625, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.07025968376547098, "epoch": 0.29582294928323427, "frac_reward_zero_std": 0.4375, "grad_norm": 0.42135199904441833, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 687377853.0, "reward": 0.40625, "reward_std": 0.24398541450500488, "rewards/simpleverify_reward/mean": 0.40625, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1736, "tools/generated_tokens": 3236.83984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1016.28125, "completions/mean_terminated_length": 1012.2353515625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06631171936169267, "epoch": 0.2959933542079366, "frac_reward_zero_std": 0.625, "grad_norm": 0.3793116807937622, "learning_rate": 1e-06, "loss": 0.0232, "num_tokens": 687702853.0, "reward": 0.49609375, "reward_std": 0.14160975813865662, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1737, "tools/generated_tokens": 2952.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1096.8984375, "completions/mean_terminated_length": 1085.62060546875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.06970110884867609, "epoch": 0.2961637591326389, "frac_reward_zero_std": 0.5625, "grad_norm": 0.6272870898246765, "learning_rate": 1e-06, "loss": 0.0405, "num_tokens": 688063403.0, "reward": 0.33984375, "reward_std": 0.13511523604393005, "rewards/simpleverify_reward/mean": 0.33984375, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1738, "tools/generated_tokens": 3584.90625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1032.62109375, "completions/mean_terminated_length": 1008.2520751953125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.07438507908955216, "epoch": 0.29633416405734125, "frac_reward_zero_std": 0.4375, "grad_norm": 0.492576539516449, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 688404730.0, "reward": 0.546875, "reward_std": 0.21632902324199677, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1739, "tools/generated_tokens": 3864.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1073.35546875, "completions/mean_terminated_length": 995.2193603515625, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.06406391225755215, "epoch": 0.2965045689820436, "frac_reward_zero_std": 0.5625, "grad_norm": 0.39404040575027466, "learning_rate": 1e-06, "loss": 0.0086, "num_tokens": 688767269.0, "reward": 0.47265625, "reward_std": 0.17143860459327698, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1740, "tools/generated_tokens": 4593.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.71875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1142.37109375, "completions/mean_terminated_length": 1138.8197021484375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.0788741479627788, "epoch": 0.2966749739067459, "frac_reward_zero_std": 0.3125, "grad_norm": 0.45827120542526245, "learning_rate": 1e-06, "loss": 0.0018, "num_tokens": 689127636.0, "reward": 0.63671875, "reward_std": 0.22380754351615906, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1741, "tools/generated_tokens": 3318.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1101.875, "completions/mean_terminated_length": 1090.6561279296875, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.06787272123619914, "epoch": 0.29684537883144824, "frac_reward_zero_std": 0.5, "grad_norm": 0.48665550351142883, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 689475828.0, "reward": 0.63671875, "reward_std": 0.18793907761573792, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1742, "tools/generated_tokens": 3341.8828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2047.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 985.7734375, "completions/mean_terminated_length": 981.61181640625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.07143229385837913, "epoch": 0.29701578375615056, "frac_reward_zero_std": 0.375, "grad_norm": 0.4373171329498291, "learning_rate": 1e-06, "loss": 0.0039, "num_tokens": 689804090.0, "reward": 0.6484375, "reward_std": 0.220298171043396, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1743, "tools/generated_tokens": 3449.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1188.56640625, "completions/mean_terminated_length": 1174.9246826171875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.06222368055023253, "epoch": 0.2971861886808529, "frac_reward_zero_std": 0.4375, "grad_norm": 0.37028542160987854, "learning_rate": 1e-06, "loss": 0.0218, "num_tokens": 690177387.0, "reward": 0.50390625, "reward_std": 0.22227820754051208, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1744, "tools/generated_tokens": 3532.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1036.5625, "completions/mean_terminated_length": 1024.5692138671875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.06929064192809165, "epoch": 0.2973565936055552, "frac_reward_zero_std": 0.5625, "grad_norm": 0.46244096755981445, "learning_rate": 1e-06, "loss": -0.033, "num_tokens": 690510379.0, "reward": 0.57421875, "reward_std": 0.17473775148391724, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1745, "tools/generated_tokens": 3100.5625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1091.86328125, "completions/mean_terminated_length": 1064.98388671875, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.07535281800664961, "epoch": 0.29752699853025755, "frac_reward_zero_std": 0.5, "grad_norm": 0.540663480758667, "learning_rate": 1e-06, "loss": -0.0327, "num_tokens": 690864536.0, "reward": 0.55078125, "reward_std": 0.21144512295722961, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1746, "tools/generated_tokens": 3595.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1010.8203125, "completions/mean_terminated_length": 977.3628540039062, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.06998799834400415, "epoch": 0.2976974034549599, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2747247517108917, "learning_rate": 1e-06, "loss": 0.0167, "num_tokens": 691201594.0, "reward": 0.42578125, "reward_std": 0.10574321448802948, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1747, "tools/generated_tokens": 3658.8125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1139.2734375, "completions/mean_terminated_length": 1098.473388671875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.06978382426314056, "epoch": 0.29786780837966215, "frac_reward_zero_std": 0.6875, "grad_norm": 0.38452887535095215, "learning_rate": 1e-06, "loss": 0.0121, "num_tokens": 691573568.0, "reward": 0.56640625, "reward_std": 0.1254390925168991, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1748, "tools/generated_tokens": 4067.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1171.2265625, "completions/mean_terminated_length": 1124.3209228515625, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.0694964500144124, "epoch": 0.2980382133043645, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3833063840866089, "learning_rate": 1e-06, "loss": 0.0597, "num_tokens": 691938218.0, "reward": 0.44921875, "reward_std": 0.14832130074501038, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1749, "tools/generated_tokens": 3723.234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.24609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1129.19921875, "completions/mean_terminated_length": 1118.304443359375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.07406028360128403, "epoch": 0.2982086182290668, "frac_reward_zero_std": 0.5, "grad_norm": 0.451852023601532, "learning_rate": 1e-06, "loss": -0.0075, "num_tokens": 692297293.0, "reward": 0.63671875, "reward_std": 0.18599742650985718, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1750, "tools/generated_tokens": 3609.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1203.94140625, "completions/mean_terminated_length": 1132.4110107421875, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.06729474826715887, "epoch": 0.29837902315376913, "frac_reward_zero_std": 0.5, "grad_norm": 0.34861573576927185, "learning_rate": 1e-06, "loss": 0.0291, "num_tokens": 692672846.0, "reward": 0.6875, "reward_std": 0.17693254351615906, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1751, "tools/generated_tokens": 3547.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.14453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2041.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1049.75, "completions/mean_terminated_length": 1049.75, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.07364571327343583, "epoch": 0.29854942807847146, "frac_reward_zero_std": 0.375, "grad_norm": 0.465381383895874, "learning_rate": 1e-06, "loss": 0.0164, "num_tokens": 693016830.0, "reward": 0.6015625, "reward_std": 0.2369977980852127, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1752, "tools/generated_tokens": 3137.7578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 974.75390625, "completions/mean_terminated_length": 974.75390625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.06759170163422823, "epoch": 0.2987198330031738, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3633814752101898, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 693346367.0, "reward": 0.55078125, "reward_std": 0.17923866212368011, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1753, "tools/generated_tokens": 3486.75390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1191.05078125, "completions/mean_terminated_length": 1170.4840087890625, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.06313716364093125, "epoch": 0.2988902379278761, "frac_reward_zero_std": 0.5, "grad_norm": 0.3714597225189209, "learning_rate": 1e-06, "loss": 0.0192, "num_tokens": 693719276.0, "reward": 0.70703125, "reward_std": 0.1630660742521286, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 1754, "tools/generated_tokens": 3583.0546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.16796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1134.14453125, "completions/mean_terminated_length": 1126.9488525390625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.06912155030295253, "epoch": 0.29906064285257844, "frac_reward_zero_std": 0.4375, "grad_norm": 0.32742252945899963, "learning_rate": 1e-06, "loss": -0.0196, "num_tokens": 694080897.0, "reward": 0.44140625, "reward_std": 0.21151351928710938, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1755, "tools/generated_tokens": 3622.1484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1208.3515625, "completions/mean_terminated_length": 1198.395263671875, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.07358960132114589, "epoch": 0.29923104777728077, "frac_reward_zero_std": 0.75, "grad_norm": 0.276409387588501, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 694454315.0, "reward": 0.51953125, "reward_std": 0.09617365896701813, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1756, "tools/generated_tokens": 2936.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1183.67578125, "completions/mean_terminated_length": 1183.67578125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.06639794330112636, "epoch": 0.2994014527019831, "frac_reward_zero_std": 0.5625, "grad_norm": 0.35287731885910034, "learning_rate": 1e-06, "loss": 0.021, "num_tokens": 694819480.0, "reward": 0.640625, "reward_std": 0.15404412150382996, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1757, "tools/generated_tokens": 2903.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.83984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1284.703125, "completions/mean_terminated_length": 1240.54541015625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.06666190479882061, "epoch": 0.2995718576266854, "frac_reward_zero_std": 0.5, "grad_norm": 0.35281673073768616, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 695221052.0, "reward": 0.390625, "reward_std": 0.17208996415138245, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1758, "tools/generated_tokens": 4004.69921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2039.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1180.3828125, "completions/mean_terminated_length": 1180.3828125, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.06347410404123366, "epoch": 0.29974226255138775, "frac_reward_zero_std": 0.6875, "grad_norm": 0.25460192561149597, "learning_rate": 1e-06, "loss": -0.0173, "num_tokens": 695591646.0, "reward": 0.4609375, "reward_std": 0.10981408506631851, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1759, "tools/generated_tokens": 3020.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2044.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1069.640625, "completions/mean_terminated_length": 1069.640625, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.07548151351511478, "epoch": 0.2999126674760901, "frac_reward_zero_std": 0.625, "grad_norm": 0.3552560806274414, "learning_rate": 1e-06, "loss": 0.0387, "num_tokens": 695934450.0, "reward": 0.66796875, "reward_std": 0.1517379879951477, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1760, "tools/generated_tokens": 2909.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2021.0, "completions/mean_length": 1115.359375, "completions/mean_terminated_length": 1065.4649658203125, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.07133043417707086, "epoch": 0.3000830724007924, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4505232870578766, "learning_rate": 1e-06, "loss": 0.0093, "num_tokens": 696296158.0, "reward": 0.5234375, "reward_std": 0.17239567637443542, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1761, "tools/generated_tokens": 4443.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1156.6875, "completions/mean_terminated_length": 1156.6875, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.06719385320320725, "epoch": 0.30025347732549473, "frac_reward_zero_std": 0.625, "grad_norm": 0.3384978473186493, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 696649006.0, "reward": 0.6875, "reward_std": 0.1281953901052475, "rewards/simpleverify_reward/mean": 0.6875, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1762, "tools/generated_tokens": 2788.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1071.5546875, "completions/mean_terminated_length": 1059.976318359375, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.06723536713980138, "epoch": 0.300423882250197, "frac_reward_zero_std": 0.25, "grad_norm": 0.49422237277030945, "learning_rate": 1e-06, "loss": 0.0327, "num_tokens": 697003836.0, "reward": 0.5, "reward_std": 0.27893561124801636, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1763, "tools/generated_tokens": 3895.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1168.15625, "completions/mean_terminated_length": 1154.1905517578125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.07576287817209959, "epoch": 0.30059428717489933, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5105679035186768, "learning_rate": 1e-06, "loss": -0.0113, "num_tokens": 697366084.0, "reward": 0.43359375, "reward_std": 0.16780208051204681, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1764, "tools/generated_tokens": 3176.16015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.98046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1157.60546875, "completions/mean_terminated_length": 1106.094970703125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.0681654626969248, "epoch": 0.30076469209960166, "frac_reward_zero_std": 0.5, "grad_norm": 0.45249611139297485, "learning_rate": 1e-06, "loss": 0.0293, "num_tokens": 697740207.0, "reward": 0.46875, "reward_std": 0.19156450033187866, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1765, "tools/generated_tokens": 3997.609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1127.1953125, "completions/mean_terminated_length": 1081.913818359375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.0712350714020431, "epoch": 0.300935097024304, "frac_reward_zero_std": 0.5, "grad_norm": 0.44048088788986206, "learning_rate": 1e-06, "loss": 0.0169, "num_tokens": 698108225.0, "reward": 0.51171875, "reward_std": 0.21755313873291016, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1766, "tools/generated_tokens": 4543.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1092.25390625, "completions/mean_terminated_length": 1073.2152099609375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.06614208593964577, "epoch": 0.3011055019490063, "frac_reward_zero_std": 0.625, "grad_norm": 0.45844268798828125, "learning_rate": 1e-06, "loss": -0.0203, "num_tokens": 698465970.0, "reward": 0.59375, "reward_std": 0.14954319596290588, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1767, "tools/generated_tokens": 3924.25390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1162.16015625, "completions/mean_terminated_length": 1140.9000244140625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07042121002450585, "epoch": 0.30127590687370864, "frac_reward_zero_std": 0.5, "grad_norm": 0.45015427470207214, "learning_rate": 1e-06, "loss": -0.0404, "num_tokens": 698837771.0, "reward": 0.5703125, "reward_std": 0.22060582041740417, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1768, "tools/generated_tokens": 3818.16015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1097.86328125, "completions/mean_terminated_length": 1090.3818359375, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.07765826489776373, "epoch": 0.30144631179841097, "frac_reward_zero_std": 0.5, "grad_norm": 0.3633653521537781, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 699189656.0, "reward": 0.5703125, "reward_std": 0.16691280901432037, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1769, "tools/generated_tokens": 3625.8671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1155.3671875, "completions/mean_terminated_length": 1148.338623046875, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.07053101528435946, "epoch": 0.3016167167231133, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2911376655101776, "learning_rate": 1e-06, "loss": -0.0139, "num_tokens": 699556406.0, "reward": 0.72265625, "reward_std": 0.16516819596290588, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1770, "tools/generated_tokens": 3675.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.23046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1172.0390625, "completions/mean_terminated_length": 1113.6417236328125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.07657679077237844, "epoch": 0.3017871216478156, "frac_reward_zero_std": 0.5, "grad_norm": 0.4389435052871704, "learning_rate": 1e-06, "loss": -0.0018, "num_tokens": 699926320.0, "reward": 0.5390625, "reward_std": 0.208757221698761, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1771, "tools/generated_tokens": 4020.03515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1061.21875, "completions/mean_terminated_length": 1057.34912109375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.07706544268876314, "epoch": 0.30195752657251795, "frac_reward_zero_std": 0.375, "grad_norm": 0.49822762608528137, "learning_rate": 1e-06, "loss": -0.0015, "num_tokens": 700269400.0, "reward": 0.61328125, "reward_std": 0.22711142897605896, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1772, "tools/generated_tokens": 3469.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1101.60546875, "completions/mean_terminated_length": 1082.7530517578125, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.07894926331937313, "epoch": 0.3021279314972203, "frac_reward_zero_std": 0.5, "grad_norm": 0.42408841848373413, "learning_rate": 1e-06, "loss": -0.0344, "num_tokens": 700626147.0, "reward": 0.41796875, "reward_std": 0.16813471913337708, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1773, "tools/generated_tokens": 4221.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1052.72265625, "completions/mean_terminated_length": 1040.9210205078125, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.07354859937913716, "epoch": 0.3022983364219226, "frac_reward_zero_std": 0.375, "grad_norm": 0.5006978511810303, "learning_rate": 1e-06, "loss": 0.0121, "num_tokens": 700983548.0, "reward": 0.52734375, "reward_std": 0.24129751324653625, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1774, "tools/generated_tokens": 4476.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1046.08203125, "completions/mean_terminated_length": 1022.0360717773438, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.07399405259639025, "epoch": 0.30246874134662494, "frac_reward_zero_std": 0.5, "grad_norm": 0.5013255476951599, "learning_rate": 1e-06, "loss": -0.0087, "num_tokens": 701333297.0, "reward": 0.515625, "reward_std": 0.18871080875396729, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1775, "tools/generated_tokens": 4142.08203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2008.0, "completions/mean_length": 1138.54296875, "completions/mean_terminated_length": 1127.7589111328125, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.06685561966150999, "epoch": 0.30263914627132726, "frac_reward_zero_std": 0.5, "grad_norm": 0.49113428592681885, "learning_rate": 1e-06, "loss": 0.0233, "num_tokens": 701697916.0, "reward": 0.578125, "reward_std": 0.1624118983745575, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1776, "tools/generated_tokens": 3498.54296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1101.7734375, "completions/mean_terminated_length": 1090.5533447265625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.0661811358295381, "epoch": 0.3028095511960296, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4624665379524231, "learning_rate": 1e-06, "loss": -0.0143, "num_tokens": 702060674.0, "reward": 0.41015625, "reward_std": 0.22579212486743927, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1777, "tools/generated_tokens": 4749.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1107.02734375, "completions/mean_terminated_length": 1064.779541015625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.07374384626746178, "epoch": 0.30297995612073186, "frac_reward_zero_std": 0.5, "grad_norm": 0.472870409488678, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 702430297.0, "reward": 0.71484375, "reward_std": 0.2086106836795807, "rewards/simpleverify_reward/mean": 0.71484375, "rewards/simpleverify_reward/std": 0.4523732364177704, "step": 1778, "tools/generated_tokens": 3907.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1051.546875, "completions/mean_terminated_length": 1043.7008056640625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.08146071061491966, "epoch": 0.3031503610454342, "frac_reward_zero_std": 0.5625, "grad_norm": 0.42284825444221497, "learning_rate": 1e-06, "loss": -0.0139, "num_tokens": 702782341.0, "reward": 0.4921875, "reward_std": 0.19430497288703918, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1779, "tools/generated_tokens": 4507.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1211.703125, "completions/mean_terminated_length": 1195.0438232421875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.062134688487276435, "epoch": 0.3033207659701365, "frac_reward_zero_std": 0.375, "grad_norm": 0.4193178415298462, "learning_rate": 1e-06, "loss": 0.0119, "num_tokens": 703175673.0, "reward": 0.60546875, "reward_std": 0.25427472591400146, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1780, "tools/generated_tokens": 4491.7109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1153.38671875, "completions/mean_terminated_length": 1146.342529296875, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.08976230374537408, "epoch": 0.30349117089483885, "frac_reward_zero_std": 0.625, "grad_norm": 0.2842784523963928, "learning_rate": 1e-06, "loss": -0.0039, "num_tokens": 703545644.0, "reward": 0.67578125, "reward_std": 0.14423459768295288, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1781, "tools/generated_tokens": 3593.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1182.09375, "completions/mean_terminated_length": 1171.826171875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.0657952178735286, "epoch": 0.3036615758195412, "frac_reward_zero_std": 0.4375, "grad_norm": 0.34768131375312805, "learning_rate": 1e-06, "loss": -0.0258, "num_tokens": 703935940.0, "reward": 0.59375, "reward_std": 0.1936618983745575, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1782, "tools/generated_tokens": 4630.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1116.5703125, "completions/mean_terminated_length": 1116.5703125, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.07101624715141952, "epoch": 0.3038319807442435, "frac_reward_zero_std": 0.6875, "grad_norm": 0.300557404756546, "learning_rate": 1e-06, "loss": 0.0098, "num_tokens": 704286486.0, "reward": 0.4453125, "reward_std": 0.1171475350856781, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1783, "tools/generated_tokens": 3452.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1059.41015625, "completions/mean_terminated_length": 1047.687744140625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.06236184109002352, "epoch": 0.30400238566894583, "frac_reward_zero_std": 0.5, "grad_norm": 0.3689342141151428, "learning_rate": 1e-06, "loss": -0.0084, "num_tokens": 704641647.0, "reward": 0.66796875, "reward_std": 0.17098368704319, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1784, "tools/generated_tokens": 3867.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1230.78125, "completions/mean_terminated_length": 1224.346435546875, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 0.07021211972460151, "epoch": 0.30417279059364816, "frac_reward_zero_std": 0.6875, "grad_norm": 0.28857752680778503, "learning_rate": 1e-06, "loss": -0.0139, "num_tokens": 705017767.0, "reward": 0.59375, "reward_std": 0.10981409251689911, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 1785, "tools/generated_tokens": 3086.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.90625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1134.19140625, "completions/mean_terminated_length": 1097.044677734375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.07305835303850472, "epoch": 0.3043431955183505, "frac_reward_zero_std": 0.375, "grad_norm": 0.5039616823196411, "learning_rate": 1e-06, "loss": 0.0045, "num_tokens": 705384808.0, "reward": 0.35546875, "reward_std": 0.22271710634231567, "rewards/simpleverify_reward/mean": 0.35546875, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1786, "tools/generated_tokens": 4566.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.67578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1118.78515625, "completions/mean_terminated_length": 1104.0357666015625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.07275776867754757, "epoch": 0.3045136004430528, "frac_reward_zero_std": 0.5625, "grad_norm": 0.37723249197006226, "learning_rate": 1e-06, "loss": 0.0267, "num_tokens": 705750225.0, "reward": 0.49609375, "reward_std": 0.16768454015254974, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1787, "tools/generated_tokens": 4054.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.43359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1138.73828125, "completions/mean_terminated_length": 1124.3056640625, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.06726943119429052, "epoch": 0.30468400536775514, "frac_reward_zero_std": 0.3125, "grad_norm": 0.46993932127952576, "learning_rate": 1e-06, "loss": 0.0064, "num_tokens": 706126286.0, "reward": 0.5234375, "reward_std": 0.2630069851875305, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1788, "tools/generated_tokens": 5066.73046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.91796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1076.21484375, "completions/mean_terminated_length": 1068.56298828125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.06391585106030107, "epoch": 0.30485441029245747, "frac_reward_zero_std": 0.5625, "grad_norm": 0.38383033871650696, "learning_rate": 1e-06, "loss": -0.0032, "num_tokens": 706488981.0, "reward": 0.65234375, "reward_std": 0.17131631076335907, "rewards/simpleverify_reward/mean": 0.65234375, "rewards/simpleverify_reward/std": 0.4771590530872345, "step": 1789, "tools/generated_tokens": 3892.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1173.70703125, "completions/mean_terminated_length": 1173.70703125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.058907060185447335, "epoch": 0.3050248152171598, "frac_reward_zero_std": 0.5625, "grad_norm": 0.2690507471561432, "learning_rate": 1e-06, "loss": -0.014, "num_tokens": 706862362.0, "reward": 0.76171875, "reward_std": 0.14547231793403625, "rewards/simpleverify_reward/mean": 0.76171875, "rewards/simpleverify_reward/std": 0.4268665909767151, "step": 1790, "tools/generated_tokens": 2989.71484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.88671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1156.94140625, "completions/mean_terminated_length": 1153.4471435546875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.06033791461959481, "epoch": 0.3051952201418621, "frac_reward_zero_std": 0.5, "grad_norm": 0.42872416973114014, "learning_rate": 1e-06, "loss": 0.0274, "num_tokens": 707237115.0, "reward": 0.671875, "reward_std": 0.19332927465438843, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1791, "tools/generated_tokens": 4020.93359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1042.6796875, "completions/mean_terminated_length": 984.5206298828125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.06575158471241593, "epoch": 0.30536562506656445, "frac_reward_zero_std": 0.4375, "grad_norm": 0.43420520424842834, "learning_rate": 1e-06, "loss": 0.0035, "num_tokens": 707583657.0, "reward": 0.5859375, "reward_std": 0.2418341487646103, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1792, "tools/generated_tokens": 3898.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1082.64453125, "completions/mean_terminated_length": 1075.0433349609375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.08223102893680334, "epoch": 0.3055360299912667, "frac_reward_zero_std": 0.5, "grad_norm": 0.4272569417953491, "learning_rate": 1e-06, "loss": 0.0238, "num_tokens": 707944958.0, "reward": 0.39453125, "reward_std": 0.21255290508270264, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1793, "tools/generated_tokens": 4642.66015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.73828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2028.0, "completions/max_terminated_length": 2028.0, "completions/mean_length": 1160.12109375, "completions/mean_terminated_length": 1160.12109375, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.07200725399889052, "epoch": 0.30570643491596905, "frac_reward_zero_std": 0.625, "grad_norm": 0.3714357018470764, "learning_rate": 1e-06, "loss": -0.0022, "num_tokens": 708318957.0, "reward": 0.43359375, "reward_std": 0.1468954086303711, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1794, "tools/generated_tokens": 3520.11328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1167.90625, "completions/mean_terminated_length": 1150.37451171875, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.059266153955832124, "epoch": 0.3058768398406714, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4418119192123413, "learning_rate": 1e-06, "loss": 0.0056, "num_tokens": 708692229.0, "reward": 0.52734375, "reward_std": 0.23859524726867676, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1795, "tools/generated_tokens": 3815.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1227.76171875, "completions/mean_terminated_length": 1197.87451171875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.06509771733544767, "epoch": 0.3060472447653737, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4828605055809021, "learning_rate": 1e-06, "loss": 0.0017, "num_tokens": 709076248.0, "reward": 0.58984375, "reward_std": 0.18362826108932495, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1796, "tools/generated_tokens": 4027.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1237.42578125, "completions/mean_terminated_length": 1214.6385498046875, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.06400364264845848, "epoch": 0.30621764969007603, "frac_reward_zero_std": 0.5, "grad_norm": 0.3279861509799957, "learning_rate": 1e-06, "loss": -0.0013, "num_tokens": 709463797.0, "reward": 0.45703125, "reward_std": 0.17845112085342407, "rewards/simpleverify_reward/mean": 0.45703125, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1797, "tools/generated_tokens": 3853.4296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.27734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1305.328125, "completions/mean_terminated_length": 1249.15966796875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.06699050683528185, "epoch": 0.30638805461477836, "frac_reward_zero_std": 0.5625, "grad_norm": 0.42040929198265076, "learning_rate": 1e-06, "loss": -0.0146, "num_tokens": 709874281.0, "reward": 0.42578125, "reward_std": 0.16625863313674927, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1798, "tools/generated_tokens": 4713.328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1200.58984375, "completions/mean_terminated_length": 1197.2667236328125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.06390750873833895, "epoch": 0.3065584595394807, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5233546495437622, "learning_rate": 1e-06, "loss": 0.0376, "num_tokens": 710259024.0, "reward": 0.6328125, "reward_std": 0.246834397315979, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1799, "tools/generated_tokens": 3872.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1086.33203125, "completions/mean_terminated_length": 1086.33203125, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.05207514972425997, "epoch": 0.306728864464183, "frac_reward_zero_std": 0.5625, "grad_norm": 0.30930477380752563, "learning_rate": 1e-06, "loss": 0.0118, "num_tokens": 710614085.0, "reward": 0.5078125, "reward_std": 0.1522855907678604, "rewards/simpleverify_reward/mean": 0.5078125, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1800, "tools/generated_tokens": 4014.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1183.79296875, "completions/mean_terminated_length": 1163.052001953125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.07273666048422456, "epoch": 0.30689926938888534, "frac_reward_zero_std": 0.5625, "grad_norm": 0.32985854148864746, "learning_rate": 1e-06, "loss": 0.0008, "num_tokens": 710976256.0, "reward": 0.55078125, "reward_std": 0.18059615790843964, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1801, "tools/generated_tokens": 3351.79296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1159.48046875, "completions/mean_terminated_length": 1134.501953125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.06356648541986942, "epoch": 0.30706967431358767, "frac_reward_zero_std": 0.625, "grad_norm": 0.36703482270240784, "learning_rate": 1e-06, "loss": 0.0021, "num_tokens": 711345915.0, "reward": 0.58203125, "reward_std": 0.18350879848003387, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1802, "tools/generated_tokens": 3631.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1122.890625, "completions/mean_terminated_length": 1104.462158203125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.06559944641776383, "epoch": 0.30724007923829, "frac_reward_zero_std": 0.5625, "grad_norm": 0.38105612993240356, "learning_rate": 1e-06, "loss": 0.0081, "num_tokens": 711714879.0, "reward": 0.421875, "reward_std": 0.1928790807723999, "rewards/simpleverify_reward/mean": 0.421875, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1803, "tools/generated_tokens": 4002.90234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2031.0, "completions/mean_length": 1245.2421875, "completions/mean_terminated_length": 1235.723388671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.06354004284366965, "epoch": 0.3074104841629923, "frac_reward_zero_std": 0.6875, "grad_norm": 0.409427285194397, "learning_rate": 1e-06, "loss": 0.0004, "num_tokens": 712098493.0, "reward": 0.65625, "reward_std": 0.1080445945262909, "rewards/simpleverify_reward/mean": 0.65625, "rewards/simpleverify_reward/std": 0.47588926553726196, "step": 1804, "tools/generated_tokens": 3437.24609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2032.0, "completions/mean_length": 1309.48046875, "completions/mean_terminated_length": 1243.485107421875, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.05931610008701682, "epoch": 0.30758088908769465, "frac_reward_zero_std": 0.5, "grad_norm": 0.4446350336074829, "learning_rate": 1e-06, "loss": 0.0193, "num_tokens": 712515032.0, "reward": 0.5859375, "reward_std": 0.19498121738433838, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1805, "tools/generated_tokens": 4293.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1158.0234375, "completions/mean_terminated_length": 1118.0653076171875, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.06627872353419662, "epoch": 0.307751294012397, "frac_reward_zero_std": 0.4375, "grad_norm": 0.489809513092041, "learning_rate": 1e-06, "loss": -0.0406, "num_tokens": 712885310.0, "reward": 0.43359375, "reward_std": 0.20103666186332703, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1806, "tools/generated_tokens": 4358.02734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1158.4140625, "completions/mean_terminated_length": 1158.4140625, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.057737802155315876, "epoch": 0.3079216989370993, "frac_reward_zero_std": 0.625, "grad_norm": 0.3039773106575012, "learning_rate": 1e-06, "loss": -0.0035, "num_tokens": 713252824.0, "reward": 0.59765625, "reward_std": 0.14501741528511047, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1807, "tools/generated_tokens": 3342.4140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1257.5859375, "completions/mean_terminated_length": 1172.052001953125, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.06845155917108059, "epoch": 0.3080921038618016, "frac_reward_zero_std": 0.5625, "grad_norm": 0.40451523661613464, "learning_rate": 1e-06, "loss": 0.0094, "num_tokens": 713660190.0, "reward": 0.4921875, "reward_std": 0.15558473765850067, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1808, "tools/generated_tokens": 4745.59765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1146.51171875, "completions/mean_terminated_length": 1132.202392578125, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.06217735982500017, "epoch": 0.3082625087865039, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3638685941696167, "learning_rate": 1e-06, "loss": -0.0036, "num_tokens": 714043041.0, "reward": 0.5234375, "reward_std": 0.23260819911956787, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1809, "tools/generated_tokens": 4018.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1207.421875, "completions/mean_terminated_length": 1207.421875, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.06039783637970686, "epoch": 0.30843291371120624, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3118876516819, "learning_rate": 1e-06, "loss": -0.0059, "num_tokens": 714423213.0, "reward": 0.67578125, "reward_std": 0.17451362311840057, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1810, "tools/generated_tokens": 3215.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.98046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1156.5625, "completions/mean_terminated_length": 1089.1470947265625, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.06534555833786726, "epoch": 0.30860331863590856, "frac_reward_zero_std": 0.5, "grad_norm": 0.4355911314487457, "learning_rate": 1e-06, "loss": -0.0179, "num_tokens": 714788333.0, "reward": 0.5546875, "reward_std": 0.20327436923980713, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1811, "tools/generated_tokens": 3788.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.28515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1353.203125, "completions/mean_terminated_length": 1281.32763671875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.05694023543037474, "epoch": 0.3087737235606109, "frac_reward_zero_std": 0.5625, "grad_norm": 0.33059704303741455, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 715214609.0, "reward": 0.52734375, "reward_std": 0.17850500345230103, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1812, "tools/generated_tokens": 4809.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2015.0, "completions/mean_length": 1307.07421875, "completions/mean_terminated_length": 1270.6351318359375, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.061854813946411014, "epoch": 0.3089441284853132, "frac_reward_zero_std": 0.75, "grad_norm": 0.24527384340763092, "learning_rate": 1e-06, "loss": 0.0133, "num_tokens": 715607396.0, "reward": 0.5, "reward_std": 0.0816391110420227, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1813, "tools/generated_tokens": 3275.0703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1157.98828125, "completions/mean_terminated_length": 1129.2781982421875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.06606793124228716, "epoch": 0.30911453341001555, "frac_reward_zero_std": 0.3125, "grad_norm": 0.6239451169967651, "learning_rate": 1e-06, "loss": 0.0252, "num_tokens": 715991121.0, "reward": 0.51953125, "reward_std": 0.2774302363395691, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1814, "tools/generated_tokens": 4445.98828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.60546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1162.08984375, "completions/mean_terminated_length": 1148.02783203125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.07045078044757247, "epoch": 0.3092849383347179, "frac_reward_zero_std": 0.375, "grad_norm": 0.5233270525932312, "learning_rate": 1e-06, "loss": 0.0187, "num_tokens": 716374552.0, "reward": 0.4765625, "reward_std": 0.28340962529182434, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1815, "tools/generated_tokens": 4738.08984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.74609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1120.2265625, "completions/mean_terminated_length": 1105.5040283203125, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.08190703578293324, "epoch": 0.3094553432594202, "frac_reward_zero_std": 0.4375, "grad_norm": 0.41888710856437683, "learning_rate": 1e-06, "loss": 0.0337, "num_tokens": 716741474.0, "reward": 0.49609375, "reward_std": 0.19916057586669922, "rewards/simpleverify_reward/mean": 0.49609375, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1816, "tools/generated_tokens": 4240.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1223.2890625, "completions/mean_terminated_length": 1213.5098876953125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.06878282455727458, "epoch": 0.30962574818412253, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3809596300125122, "learning_rate": 1e-06, "loss": -0.0005, "num_tokens": 717133644.0, "reward": 0.4765625, "reward_std": 0.12466736882925034, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1817, "tools/generated_tokens": 4495.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1097.125, "completions/mean_terminated_length": 1070.3935546875, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.0638113294262439, "epoch": 0.30979615310882486, "frac_reward_zero_std": 0.5, "grad_norm": 0.4144458472728729, "learning_rate": 1e-06, "loss": 0.0389, "num_tokens": 717490988.0, "reward": 0.55859375, "reward_std": 0.19188131392002106, "rewards/simpleverify_reward/mean": 0.55859375, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1818, "tools/generated_tokens": 3841.12890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.33984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1256.89453125, "completions/mean_terminated_length": 1224.7357177734375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.06858780211769044, "epoch": 0.3099665580335272, "frac_reward_zero_std": 0.375, "grad_norm": 0.5148463845252991, "learning_rate": 1e-06, "loss": 0.0322, "num_tokens": 717888513.0, "reward": 0.57421875, "reward_std": 0.2468073070049286, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1819, "tools/generated_tokens": 4432.890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1231.21484375, "completions/mean_terminated_length": 1191.0450439453125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.06323532178066671, "epoch": 0.3101369629582295, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4007709324359894, "learning_rate": 1e-06, "loss": -0.0083, "num_tokens": 718284008.0, "reward": 0.54296875, "reward_std": 0.1757626235485077, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1820, "tools/generated_tokens": 4655.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1286.8203125, "completions/mean_terminated_length": 1236.0750732421875, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.06365184881724417, "epoch": 0.31030736788293184, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4121614098548889, "learning_rate": 1e-06, "loss": 0.0229, "num_tokens": 718697338.0, "reward": 0.4609375, "reward_std": 0.2091582715511322, "rewards/simpleverify_reward/mean": 0.4609375, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1821, "tools/generated_tokens": 4934.82421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.78125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1275.80859375, "completions/mean_terminated_length": 1203.2137451171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.06262039439752698, "epoch": 0.31047777280763417, "frac_reward_zero_std": 0.6875, "grad_norm": 0.38320380449295044, "learning_rate": 1e-06, "loss": 0.0028, "num_tokens": 719113705.0, "reward": 0.46484375, "reward_std": 0.13226625323295593, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1822, "tools/generated_tokens": 4715.828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1150.46484375, "completions/mean_terminated_length": 1082.5841064453125, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.062300048768520355, "epoch": 0.31064817773233644, "frac_reward_zero_std": 0.5, "grad_norm": 0.44506093859672546, "learning_rate": 1e-06, "loss": -0.0296, "num_tokens": 719496288.0, "reward": 0.42578125, "reward_std": 0.20070403814315796, "rewards/simpleverify_reward/mean": 0.42578125, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1823, "tools/generated_tokens": 4598.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1186.703125, "completions/mean_terminated_length": 1176.4901123046875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.06392571423202753, "epoch": 0.31081858265703877, "frac_reward_zero_std": 0.5, "grad_norm": 0.37330999970436096, "learning_rate": 1e-06, "loss": -0.0135, "num_tokens": 719869876.0, "reward": 0.62890625, "reward_std": 0.17946279048919678, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1824, "tools/generated_tokens": 3434.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1150.30859375, "completions/mean_terminated_length": 1106.1597900390625, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.07333901803940535, "epoch": 0.3109889875817411, "frac_reward_zero_std": 0.5, "grad_norm": 1.0655543804168701, "learning_rate": 1e-06, "loss": 0.003, "num_tokens": 720245331.0, "reward": 0.5, "reward_std": 0.215584397315979, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1825, "tools/generated_tokens": 4478.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1188.625, "completions/mean_terminated_length": 1115.8050537109375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.07243580906651914, "epoch": 0.3111593925064434, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5871554613113403, "learning_rate": 1e-06, "loss": 0.0662, "num_tokens": 720632963.0, "reward": 0.3984375, "reward_std": 0.21245741844177246, "rewards/simpleverify_reward/mean": 0.3984375, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1826, "tools/generated_tokens": 5244.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.98046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1228.89453125, "completions/mean_terminated_length": 1132.31884765625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.06668501649983227, "epoch": 0.31132979743114575, "frac_reward_zero_std": 0.5, "grad_norm": 0.3617124557495117, "learning_rate": 1e-06, "loss": -0.0226, "num_tokens": 721033000.0, "reward": 0.5703125, "reward_std": 0.1712368279695511, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1827, "tools/generated_tokens": 4716.89453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1167.671875, "completions/mean_terminated_length": 1150.135498046875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.07829150976613164, "epoch": 0.3115002023558481, "frac_reward_zero_std": 0.6875, "grad_norm": 0.38928020000457764, "learning_rate": 1e-06, "loss": -0.0073, "num_tokens": 721400868.0, "reward": 0.64453125, "reward_std": 0.12037044763565063, "rewards/simpleverify_reward/mean": 0.64453125, "rewards/simpleverify_reward/std": 0.4795927405357361, "step": 1828, "tools/generated_tokens": 3303.671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1192.9140625, "completions/mean_terminated_length": 1165.33056640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.06864371546544135, "epoch": 0.3116706072805504, "frac_reward_zero_std": 0.4375, "grad_norm": 0.47027021646499634, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 721771950.0, "reward": 0.6328125, "reward_std": 0.21746239066123962, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1829, "tools/generated_tokens": 3832.9140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1209.5625, "completions/mean_terminated_length": 1098.2655029296875, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.06595775508321822, "epoch": 0.31184101220525273, "frac_reward_zero_std": 0.3125, "grad_norm": 0.6769089102745056, "learning_rate": 1e-06, "loss": 0.0363, "num_tokens": 722164862.0, "reward": 0.41015625, "reward_std": 0.2777227461338043, "rewards/simpleverify_reward/mean": 0.41015625, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1830, "tools/generated_tokens": 5473.55859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 2.08203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 979.35546875, "completions/mean_terminated_length": 979.35546875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.0694581896532327, "epoch": 0.31201141712995506, "frac_reward_zero_std": 0.5625, "grad_norm": 0.41308146715164185, "learning_rate": 1e-06, "loss": -0.0133, "num_tokens": 722495385.0, "reward": 0.67578125, "reward_std": 0.16164490580558777, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1831, "tools/generated_tokens": 3891.35546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1184.1484375, "completions/mean_terminated_length": 1141.663818359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.07942187739536166, "epoch": 0.3121818220546574, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3144334554672241, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 722878799.0, "reward": 0.54296875, "reward_std": 0.10881631821393967, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1832, "tools/generated_tokens": 3920.140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1128.58203125, "completions/mean_terminated_length": 1124.9765625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.07113590324297547, "epoch": 0.3123522269793597, "frac_reward_zero_std": 0.5, "grad_norm": 0.36654749512672424, "learning_rate": 1e-06, "loss": -0.0149, "num_tokens": 723237028.0, "reward": 0.7109375, "reward_std": 0.18330952525138855, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1833, "tools/generated_tokens": 3256.57421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1097.359375, "completions/mean_terminated_length": 1078.42236328125, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.08589660143479705, "epoch": 0.31252263190406204, "frac_reward_zero_std": 0.625, "grad_norm": 0.5161654949188232, "learning_rate": 1e-06, "loss": 0.0033, "num_tokens": 723596480.0, "reward": 0.44140625, "reward_std": 0.14843884110450745, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1834, "tools/generated_tokens": 3913.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1147.59765625, "completions/mean_terminated_length": 1140.5118408203125, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.0706815430894494, "epoch": 0.31269303682876437, "frac_reward_zero_std": 0.25, "grad_norm": 0.45688676834106445, "learning_rate": 1e-06, "loss": -0.0359, "num_tokens": 723963177.0, "reward": 0.7109375, "reward_std": 0.2653988301753998, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1835, "tools/generated_tokens": 3651.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1172.234375, "completions/mean_terminated_length": 1106.0042724609375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.09422451537102461, "epoch": 0.3128634417534667, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4262969493865967, "learning_rate": 1e-06, "loss": -0.0019, "num_tokens": 724353205.0, "reward": 0.44140625, "reward_std": 0.1943160742521286, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1836, "tools/generated_tokens": 4356.2421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1162.2890625, "completions/mean_terminated_length": 1130.0162353515625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.07726037874817848, "epoch": 0.313033846678169, "frac_reward_zero_std": 0.5, "grad_norm": 0.4145093560218811, "learning_rate": 1e-06, "loss": 0.0085, "num_tokens": 724730191.0, "reward": 0.6328125, "reward_std": 0.2079564929008484, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1837, "tools/generated_tokens": 3930.28515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1149.609375, "completions/mean_terminated_length": 1131.713134765625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07298355433158576, "epoch": 0.3132042516028713, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6122732758522034, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 725099595.0, "reward": 0.58203125, "reward_std": 0.24635712802410126, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1838, "tools/generated_tokens": 4173.609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1126.34375, "completions/mean_terminated_length": 1115.4150390625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.07623963709920645, "epoch": 0.3133746565275736, "frac_reward_zero_std": 0.375, "grad_norm": 0.6322006583213806, "learning_rate": 1e-06, "loss": 0.0109, "num_tokens": 725460195.0, "reward": 0.73046875, "reward_std": 0.260769248008728, "rewards/simpleverify_reward/mean": 0.73046875, "rewards/simpleverify_reward/std": 0.44458550214767456, "step": 1839, "tools/generated_tokens": 3294.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1027.03515625, "completions/mean_terminated_length": 1023.0314331054688, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.07984261540696025, "epoch": 0.31354506145227595, "frac_reward_zero_std": 0.5, "grad_norm": 0.481623113155365, "learning_rate": 1e-06, "loss": -0.0312, "num_tokens": 725808396.0, "reward": 0.48046875, "reward_std": 0.17836952209472656, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1840, "tools/generated_tokens": 3979.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2025.0, "completions/mean_length": 1124.48046875, "completions/mean_terminated_length": 1098.51806640625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.08681248081848025, "epoch": 0.3137154663769783, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5559415221214294, "learning_rate": 1e-06, "loss": -0.0173, "num_tokens": 726158999.0, "reward": 0.4765625, "reward_std": 0.1681618094444275, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1841, "tools/generated_tokens": 3804.46875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.30859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1096.7734375, "completions/mean_terminated_length": 1093.043212890625, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.08955360995605588, "epoch": 0.3138858713016806, "frac_reward_zero_std": 0.6875, "grad_norm": 0.35538986325263977, "learning_rate": 1e-06, "loss": -0.0067, "num_tokens": 726516205.0, "reward": 0.59765625, "reward_std": 0.1076192855834961, "rewards/simpleverify_reward/mean": 0.59765625, "rewards/simpleverify_reward/std": 0.4913311004638672, "step": 1842, "tools/generated_tokens": 3032.76171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1162.5234375, "completions/mean_terminated_length": 1155.5511474609375, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.08671532571315765, "epoch": 0.31405627622638294, "frac_reward_zero_std": 0.5625, "grad_norm": 0.580024242401123, "learning_rate": 1e-06, "loss": -0.0038, "num_tokens": 726884387.0, "reward": 0.3515625, "reward_std": 0.14886415004730225, "rewards/simpleverify_reward/mean": 0.3515625, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1843, "tools/generated_tokens": 3874.52734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.32421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1175.48828125, "completions/mean_terminated_length": 1143.6964111328125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.07595740770921111, "epoch": 0.31422668115108526, "frac_reward_zero_std": 0.625, "grad_norm": 0.3375977873802185, "learning_rate": 1e-06, "loss": 0.003, "num_tokens": 727253680.0, "reward": 0.4296875, "reward_std": 0.17505928874015808, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1844, "tools/generated_tokens": 3151.48828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1124.4296875, "completions/mean_terminated_length": 1113.478271484375, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07768685929477215, "epoch": 0.3143970860757876, "frac_reward_zero_std": 0.5, "grad_norm": 0.6905287504196167, "learning_rate": 1e-06, "loss": -0.008, "num_tokens": 727611374.0, "reward": 0.66796875, "reward_std": 0.20651951432228088, "rewards/simpleverify_reward/mean": 0.66796875, "rewards/simpleverify_reward/std": 0.4718646705150604, "step": 1845, "tools/generated_tokens": 3580.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1101.24609375, "completions/mean_terminated_length": 1086.21826171875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.0827366509474814, "epoch": 0.3145674910004899, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6033199429512024, "learning_rate": 1e-06, "loss": 0.0443, "num_tokens": 727961165.0, "reward": 0.44921875, "reward_std": 0.23018452525138855, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1846, "tools/generated_tokens": 3805.25, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1069.4453125, "completions/mean_terminated_length": 1049.9561767578125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.08323518419638276, "epoch": 0.31473789592519225, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5959662795066833, "learning_rate": 1e-06, "loss": 0.0543, "num_tokens": 728306527.0, "reward": 0.47265625, "reward_std": 0.18497256934642792, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1847, "tools/generated_tokens": 3717.453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1094.01171875, "completions/mean_terminated_length": 1086.5, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.07684739213436842, "epoch": 0.3149083008498946, "frac_reward_zero_std": 0.625, "grad_norm": 0.46415066719055176, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 728657250.0, "reward": 0.6015625, "reward_std": 0.14358042180538177, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1848, "tools/generated_tokens": 3422.0234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1065.2578125, "completions/mean_terminated_length": 1037.6304931640625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.07600807538256049, "epoch": 0.3150787057745969, "frac_reward_zero_std": 0.375, "grad_norm": 0.586067795753479, "learning_rate": 1e-06, "loss": 0.0027, "num_tokens": 729001892.0, "reward": 0.609375, "reward_std": 0.2715497612953186, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1849, "tools/generated_tokens": 3801.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1093.140625, "completions/mean_terminated_length": 1062.3426513671875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.079100183211267, "epoch": 0.31524911069929923, "frac_reward_zero_std": 0.875, "grad_norm": 0.15220609307289124, "learning_rate": 1e-06, "loss": -0.0174, "num_tokens": 729344520.0, "reward": 0.44921875, "reward_std": 0.04357584938406944, "rewards/simpleverify_reward/mean": 0.44921875, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1850, "tools/generated_tokens": 3517.18359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1074.0703125, "completions/mean_terminated_length": 1062.5257568359375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.08009379403665662, "epoch": 0.31541951562400156, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6354038715362549, "learning_rate": 1e-06, "loss": -0.0142, "num_tokens": 729701658.0, "reward": 0.51171875, "reward_std": 0.21896496415138245, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1851, "tools/generated_tokens": 3922.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1181.33203125, "completions/mean_terminated_length": 1164.0677490234375, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.08115925779566169, "epoch": 0.3155899205487039, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4849849343299866, "learning_rate": 1e-06, "loss": -0.0046, "num_tokens": 730070895.0, "reward": 0.62890625, "reward_std": 0.21064913272857666, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1852, "tools/generated_tokens": 3669.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21484375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 984.3984375, "completions/mean_terminated_length": 984.3984375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.07141777500510216, "epoch": 0.31576032547340616, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5632069706916809, "learning_rate": 1e-06, "loss": -0.0357, "num_tokens": 730398757.0, "reward": 0.51171875, "reward_std": 0.21071279048919678, "rewards/simpleverify_reward/mean": 0.51171875, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1853, "tools/generated_tokens": 3656.43359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1087.640625, "completions/mean_terminated_length": 1087.640625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.07729189191013575, "epoch": 0.3159307303981085, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5347505807876587, "learning_rate": 1e-06, "loss": -0.0222, "num_tokens": 730750441.0, "reward": 0.62890625, "reward_std": 0.23832820355892181, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1854, "tools/generated_tokens": 3423.65625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1035.015625, "completions/mean_terminated_length": 1018.9405517578125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.07995935110375285, "epoch": 0.3161011353228108, "frac_reward_zero_std": 0.5, "grad_norm": 0.5145865082740784, "learning_rate": 1e-06, "loss": 0.0083, "num_tokens": 731089309.0, "reward": 0.56640625, "reward_std": 0.1829879879951477, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1855, "tools/generated_tokens": 4107.03125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2045.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1137.2734375, "completions/mean_terminated_length": 1137.2734375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.07570744305849075, "epoch": 0.31627154024751314, "frac_reward_zero_std": 0.5, "grad_norm": 0.42338430881500244, "learning_rate": 1e-06, "loss": -0.0056, "num_tokens": 731450323.0, "reward": 0.703125, "reward_std": 0.19343777000904083, "rewards/simpleverify_reward/mean": 0.703125, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 1856, "tools/generated_tokens": 3329.26953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1068.69140625, "completions/mean_terminated_length": 1064.85107421875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.08028119523078203, "epoch": 0.31644194517221547, "frac_reward_zero_std": 0.5625, "grad_norm": 0.7303881049156189, "learning_rate": 1e-06, "loss": -0.0125, "num_tokens": 731792964.0, "reward": 0.609375, "reward_std": 0.13896197080612183, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1857, "tools/generated_tokens": 3508.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2040.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1044.80078125, "completions/mean_terminated_length": 1044.80078125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.06822781031951308, "epoch": 0.3166123500969178, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4398970901966095, "learning_rate": 1e-06, "loss": -0.0133, "num_tokens": 732123233.0, "reward": 0.640625, "reward_std": 0.17561796307563782, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1858, "tools/generated_tokens": 2972.8046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.94140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 988.27734375, "completions/mean_terminated_length": 984.1216430664062, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.08466403372585773, "epoch": 0.3167827550216201, "frac_reward_zero_std": 0.5, "grad_norm": 0.4486192464828491, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 732449064.0, "reward": 0.41796875, "reward_std": 0.23600003123283386, "rewards/simpleverify_reward/mean": 0.41796875, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1859, "tools/generated_tokens": 3684.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2026.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 1060.70703125, "completions/mean_terminated_length": 1060.70703125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.08343060361221433, "epoch": 0.31695315994632245, "frac_reward_zero_std": 0.375, "grad_norm": 0.5599002838134766, "learning_rate": 1e-06, "loss": 0.0003, "num_tokens": 732783261.0, "reward": 0.60546875, "reward_std": 0.25528639554977417, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1860, "tools/generated_tokens": 3060.70703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1054.36328125, "completions/mean_terminated_length": 1050.4666748046875, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.075624561868608, "epoch": 0.3171235648710248, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5863634347915649, "learning_rate": 1e-06, "loss": -0.018, "num_tokens": 733106170.0, "reward": 0.6328125, "reward_std": 0.16901493072509766, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1861, "tools/generated_tokens": 2830.375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1099.17578125, "completions/mean_terminated_length": 1087.9249267578125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.07974620396271348, "epoch": 0.3172939697957271, "frac_reward_zero_std": 0.375, "grad_norm": 0.5021299123764038, "learning_rate": 1e-06, "loss": -0.0448, "num_tokens": 733458167.0, "reward": 0.5234375, "reward_std": 0.23966141045093536, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1862, "tools/generated_tokens": 3891.1796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1025.68359375, "completions/mean_terminated_length": 1021.674560546875, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.08468664903193712, "epoch": 0.31746437472042943, "frac_reward_zero_std": 0.625, "grad_norm": 0.5460114479064941, "learning_rate": 1e-06, "loss": -0.0073, "num_tokens": 733792838.0, "reward": 0.57421875, "reward_std": 0.14501741528511047, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1863, "tools/generated_tokens": 3521.68359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1030.5859375, "completions/mean_terminated_length": 1006.1680297851562, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "entropy": 0.08186390763148665, "epoch": 0.31763477964513176, "frac_reward_zero_std": 0.375, "grad_norm": 0.5911335349082947, "learning_rate": 1e-06, "loss": 0.001, "num_tokens": 734131548.0, "reward": 0.51953125, "reward_std": 0.2398606836795807, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1864, "tools/generated_tokens": 3070.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.99609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1024.4453125, "completions/mean_terminated_length": 1024.4453125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08879612106829882, "epoch": 0.3178051845698341, "frac_reward_zero_std": 0.375, "grad_norm": 0.5424478650093079, "learning_rate": 1e-06, "loss": -0.0185, "num_tokens": 734457342.0, "reward": 0.60546875, "reward_std": 0.24647468328475952, "rewards/simpleverify_reward/mean": 0.60546875, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1865, "tools/generated_tokens": 3440.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1076.08984375, "completions/mean_terminated_length": 1072.2784423828125, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.0898197959177196, "epoch": 0.3179755894945364, "frac_reward_zero_std": 0.625, "grad_norm": 0.5117276906967163, "learning_rate": 1e-06, "loss": -0.0057, "num_tokens": 734805045.0, "reward": 0.5390625, "reward_std": 0.1594453752040863, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1866, "tools/generated_tokens": 3460.0859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1005.21484375, "completions/mean_terminated_length": 992.8538208007812, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.09374799206852913, "epoch": 0.31814599441923874, "frac_reward_zero_std": 0.6875, "grad_norm": 0.5604315400123596, "learning_rate": 1e-06, "loss": -0.0069, "num_tokens": 735133564.0, "reward": 0.62890625, "reward_std": 0.13039490580558777, "rewards/simpleverify_reward/mean": 0.62890625, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1867, "tools/generated_tokens": 3261.2265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1030.80078125, "completions/mean_terminated_length": 1006.3880615234375, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.07869214005768299, "epoch": 0.318316399343941, "frac_reward_zero_std": 0.5, "grad_norm": 0.6041160225868225, "learning_rate": 1e-06, "loss": -0.0226, "num_tokens": 735474041.0, "reward": 0.4140625, "reward_std": 0.22651143372058868, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1868, "tools/generated_tokens": 3902.80078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.40234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1181.81640625, "completions/mean_terminated_length": 1174.99609375, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.08387887850403786, "epoch": 0.31848680426864334, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2582102119922638, "learning_rate": 1e-06, "loss": 0.0106, "num_tokens": 735831770.0, "reward": 0.68359375, "reward_std": 0.12082535773515701, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1869, "tools/generated_tokens": 2701.84765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 957.2890625, "completions/mean_terminated_length": 948.7008056640625, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.08054040465503931, "epoch": 0.31865720919334567, "frac_reward_zero_std": 0.375, "grad_norm": 0.6053634285926819, "learning_rate": 1e-06, "loss": -0.0309, "num_tokens": 736158500.0, "reward": 0.55078125, "reward_std": 0.222492977976799, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1870, "tools/generated_tokens": 3797.2890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.38671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1025.09375, "completions/mean_terminated_length": 1021.0823974609375, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.07968076784163713, "epoch": 0.318827614118048, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4506107270717621, "learning_rate": 1e-06, "loss": -0.0029, "num_tokens": 736492108.0, "reward": 0.578125, "reward_std": 0.21666640043258667, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1871, "tools/generated_tokens": 3289.09375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.10546875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1088.3046875, "completions/mean_terminated_length": 1084.541259765625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.08380438480526209, "epoch": 0.3189980190427503, "frac_reward_zero_std": 0.4375, "grad_norm": 0.52580726146698, "learning_rate": 1e-06, "loss": -0.0136, "num_tokens": 736841258.0, "reward": 0.4453125, "reward_std": 0.19667133688926697, "rewards/simpleverify_reward/mean": 0.4453125, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1872, "tools/generated_tokens": 3416.30859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 971.40625, "completions/mean_terminated_length": 971.40625, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.08729689614847302, "epoch": 0.31916842396745265, "frac_reward_zero_std": 0.625, "grad_norm": 0.3929925262928009, "learning_rate": 1e-06, "loss": -0.0079, "num_tokens": 737160882.0, "reward": 0.6484375, "reward_std": 0.12159235030412674, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1873, "tools/generated_tokens": 3171.3984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.07421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 960.48046875, "completions/mean_terminated_length": 947.5850219726562, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08530768612399697, "epoch": 0.319338828892155, "frac_reward_zero_std": 0.375, "grad_norm": 0.5508963465690613, "learning_rate": 1e-06, "loss": -0.006, "num_tokens": 737477165.0, "reward": 0.4765625, "reward_std": 0.22885413467884064, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1874, "tools/generated_tokens": 3808.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1138.83203125, "completions/mean_terminated_length": 1135.2667236328125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08897597016766667, "epoch": 0.3195092338168573, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5205336809158325, "learning_rate": 1e-06, "loss": 0.0203, "num_tokens": 737831138.0, "reward": 0.5390625, "reward_std": 0.20544488728046417, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1875, "tools/generated_tokens": 3154.83203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1012.359375, "completions/mean_terminated_length": 995.9207153320312, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.08640373777598143, "epoch": 0.31967963874155964, "frac_reward_zero_std": 0.6875, "grad_norm": 0.29894745349884033, "learning_rate": 1e-06, "loss": -0.0153, "num_tokens": 738150110.0, "reward": 0.37109375, "reward_std": 0.10881631821393967, "rewards/simpleverify_reward/mean": 0.37109375, "rewards/simpleverify_reward/std": 0.48404383659362793, "step": 1876, "tools/generated_tokens": 3204.37109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2047.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1010.43359375, "completions/mean_terminated_length": 1010.43359375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.087667943444103, "epoch": 0.31985004366626196, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4083718955516815, "learning_rate": 1e-06, "loss": -0.017, "num_tokens": 738475581.0, "reward": 0.51953125, "reward_std": 0.14656277000904083, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1877, "tools/generated_tokens": 3234.44140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2033.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1152.6171875, "completions/mean_terminated_length": 1152.6171875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.08081592572852969, "epoch": 0.3200204485909643, "frac_reward_zero_std": 0.625, "grad_norm": 0.30365705490112305, "learning_rate": 1e-06, "loss": -0.0061, "num_tokens": 738829147.0, "reward": 0.4921875, "reward_std": 0.11179865896701813, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1878, "tools/generated_tokens": 3016.6171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.91015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 844.5, "completions/mean_terminated_length": 844.5, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.08628372754901648, "epoch": 0.3201908535156666, "frac_reward_zero_std": 0.625, "grad_norm": 0.3716977536678314, "learning_rate": 1e-06, "loss": 0.0091, "num_tokens": 739116987.0, "reward": 0.6015625, "reward_std": 0.13423693180084229, "rewards/simpleverify_reward/mean": 0.6015625, "rewards/simpleverify_reward/std": 0.4905354380607605, "step": 1879, "tools/generated_tokens": 3436.5078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2035.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1044.40625, "completions/mean_terminated_length": 1044.40625, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.08080336451530457, "epoch": 0.32036125844036895, "frac_reward_zero_std": 0.625, "grad_norm": 0.32263660430908203, "learning_rate": 1e-06, "loss": -0.0174, "num_tokens": 739440707.0, "reward": 0.7109375, "reward_std": 0.137538880109787, "rewards/simpleverify_reward/mean": 0.7109375, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1880, "tools/generated_tokens": 2412.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1099.15234375, "completions/mean_terminated_length": 1080.2509765625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.07835329324007034, "epoch": 0.3205316633650713, "frac_reward_zero_std": 0.75, "grad_norm": 0.2898080050945282, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 739783082.0, "reward": 0.4921875, "reward_std": 0.09477485716342926, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1881, "tools/generated_tokens": 2939.15234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2041.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 976.48828125, "completions/mean_terminated_length": 976.48828125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.0879957266151905, "epoch": 0.3207020682897736, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4570309519767761, "learning_rate": 1e-06, "loss": -0.0402, "num_tokens": 740114855.0, "reward": 0.47265625, "reward_std": 0.23161043226718903, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1882, "tools/generated_tokens": 4096.4921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2036.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 978.37890625, "completions/mean_terminated_length": 978.37890625, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.09804452070966363, "epoch": 0.3208724732144759, "frac_reward_zero_std": 0.375, "grad_norm": 0.7990590929985046, "learning_rate": 1e-06, "loss": 0.0049, "num_tokens": 740428552.0, "reward": 0.625, "reward_std": 0.21951062977313995, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1883, "tools/generated_tokens": 3106.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2041.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 940.2734375, "completions/mean_terminated_length": 940.2734375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0910124909132719, "epoch": 0.3210428781391782, "frac_reward_zero_std": 0.5625, "grad_norm": 0.508712649345398, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 740745534.0, "reward": 0.375, "reward_std": 0.1392945945262909, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1884, "tools/generated_tokens": 3932.28125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2043.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1112.015625, "completions/mean_terminated_length": 1112.015625, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.09354925714433193, "epoch": 0.32121328306388053, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5241032242774963, "learning_rate": 1e-06, "loss": -0.0223, "num_tokens": 741095282.0, "reward": 0.69140625, "reward_std": 0.16448915004730225, "rewards/simpleverify_reward/mean": 0.69140625, "rewards/simpleverify_reward/std": 0.46281787753105164, "step": 1885, "tools/generated_tokens": 3032.01953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2010.0, "completions/mean_length": 1000.67578125, "completions/mean_terminated_length": 988.2569580078125, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.09288890659809113, "epoch": 0.32138368798858286, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5460103154182434, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 741425375.0, "reward": 0.4375, "reward_std": 0.14766712486743927, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1886, "tools/generated_tokens": 3440.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1033.41796875, "completions/mean_terminated_length": 1025.4290771484375, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.08683209540322423, "epoch": 0.3215540929132852, "frac_reward_zero_std": 0.75, "grad_norm": 0.4358604848384857, "learning_rate": 1e-06, "loss": -0.0299, "num_tokens": 741754154.0, "reward": 0.4375, "reward_std": 0.09108919650316238, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1887, "tools/generated_tokens": 3137.41015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.02734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2041.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 951.375, "completions/mean_terminated_length": 951.375, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.09202912822365761, "epoch": 0.3217244978379875, "frac_reward_zero_std": 0.5625, "grad_norm": 0.7753656506538391, "learning_rate": 1e-06, "loss": -0.0157, "num_tokens": 742070442.0, "reward": 0.5859375, "reward_std": 0.18935108184814453, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1888, "tools/generated_tokens": 3175.37890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2043.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 994.1953125, "completions/mean_terminated_length": 994.1953125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.08692240342497826, "epoch": 0.32189490276268984, "frac_reward_zero_std": 0.6875, "grad_norm": 0.41210606694221497, "learning_rate": 1e-06, "loss": -0.016, "num_tokens": 742386332.0, "reward": 0.625, "reward_std": 0.10519562661647797, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1889, "tools/generated_tokens": 2802.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.8828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2039.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 901.98046875, "completions/mean_terminated_length": 901.98046875, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.08919832250103354, "epoch": 0.32206530768739217, "frac_reward_zero_std": 0.5, "grad_norm": 0.7692736387252808, "learning_rate": 1e-06, "loss": 0.0056, "num_tokens": 742685031.0, "reward": 0.515625, "reward_std": 0.18330952525138855, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1890, "tools/generated_tokens": 3229.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 887.6171875, "completions/mean_terminated_length": 887.6171875, "completions/min_length": 5.0, "completions/min_terminated_length": 5.0, "entropy": 0.09712161170318723, "epoch": 0.3222357126120945, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3349357843399048, "learning_rate": 1e-06, "loss": -0.005, "num_tokens": 742979973.0, "reward": 0.5, "reward_std": 0.09814241528511047, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1891, "tools/generated_tokens": 2687.609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.87890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2026.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 999.71875, "completions/mean_terminated_length": 999.71875, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.09544813074171543, "epoch": 0.3224061175367968, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5695022940635681, "learning_rate": 1e-06, "loss": -0.0259, "num_tokens": 743299757.0, "reward": 0.57421875, "reward_std": 0.21720924973487854, "rewards/simpleverify_reward/mean": 0.57421875, "rewards/simpleverify_reward/std": 0.49542948603630066, "step": 1892, "tools/generated_tokens": 3311.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 985.57421875, "completions/mean_terminated_length": 977.2086791992188, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.09465279104188085, "epoch": 0.32257652246149915, "frac_reward_zero_std": 0.6875, "grad_norm": 0.2789136469364166, "learning_rate": 1e-06, "loss": -0.0442, "num_tokens": 743630848.0, "reward": 0.5703125, "reward_std": 0.12820856273174286, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1893, "tools/generated_tokens": 3553.5703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.25390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2026.0, "completions/max_terminated_length": 2026.0, "completions/mean_length": 978.1328125, "completions/mean_terminated_length": 978.1328125, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08593334024772048, "epoch": 0.3227469273862015, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6035968065261841, "learning_rate": 1e-06, "loss": 0.0126, "num_tokens": 743951362.0, "reward": 0.4765625, "reward_std": 0.25343742966651917, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1894, "tools/generated_tokens": 3338.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1091.9609375, "completions/mean_terminated_length": 1065.0843505859375, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.09187000058591366, "epoch": 0.3229173323109038, "frac_reward_zero_std": 0.375, "grad_norm": 0.6147284507751465, "learning_rate": 1e-06, "loss": -0.0223, "num_tokens": 744311544.0, "reward": 0.50390625, "reward_std": 0.2337125539779663, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1895, "tools/generated_tokens": 4203.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1013.359375, "completions/mean_terminated_length": 1009.302001953125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.08808405883610249, "epoch": 0.32308773723560613, "frac_reward_zero_std": 0.5625, "grad_norm": 0.46973732113838196, "learning_rate": 1e-06, "loss": -0.064, "num_tokens": 744629156.0, "reward": 0.63671875, "reward_std": 0.16956061124801636, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1896, "tools/generated_tokens": 3101.359375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 983.10546875, "completions/mean_terminated_length": 978.929443359375, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.09175283648073673, "epoch": 0.32325814216030846, "frac_reward_zero_std": 0.625, "grad_norm": 0.5142419934272766, "learning_rate": 1e-06, "loss": 0.0102, "num_tokens": 744953087.0, "reward": 0.50390625, "reward_std": 0.1413172334432602, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1897, "tools/generated_tokens": 3047.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2033.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 972.58203125, "completions/mean_terminated_length": 972.58203125, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "entropy": 0.09492975566536188, "epoch": 0.32342854708501073, "frac_reward_zero_std": 0.5, "grad_norm": 0.5574467778205872, "learning_rate": 1e-06, "loss": 0.0111, "num_tokens": 745272724.0, "reward": 0.375, "reward_std": 0.18815404176712036, "rewards/simpleverify_reward/mean": 0.375, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1898, "tools/generated_tokens": 3332.578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1064.8515625, "completions/mean_terminated_length": 1064.8515625, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.08080014819279313, "epoch": 0.32359895200971306, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4713577926158905, "learning_rate": 1e-06, "loss": 0.0078, "num_tokens": 745605422.0, "reward": 0.74609375, "reward_std": 0.1744433045387268, "rewards/simpleverify_reward/mean": 0.74609375, "rewards/simpleverify_reward/std": 0.4360972046852112, "step": 1899, "tools/generated_tokens": 2584.86328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.7421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 914.5390625, "completions/mean_terminated_length": 905.6141967773438, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.08830131078138947, "epoch": 0.3237693569344154, "frac_reward_zero_std": 0.5, "grad_norm": 0.5203280448913574, "learning_rate": 1e-06, "loss": -0.0295, "num_tokens": 745912040.0, "reward": 0.5, "reward_std": 0.21047285199165344, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1900, "tools/generated_tokens": 3322.55078125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1040.7734375, "completions/mean_terminated_length": 1032.842529296875, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.0874418281018734, "epoch": 0.3239397618591177, "frac_reward_zero_std": 0.375, "grad_norm": 0.7632402181625366, "learning_rate": 1e-06, "loss": -0.0244, "num_tokens": 746251070.0, "reward": 0.6484375, "reward_std": 0.200038880109787, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1901, "tools/generated_tokens": 3352.76953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 980.8828125, "completions/mean_terminated_length": 980.8828125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.09379322407767177, "epoch": 0.32411016678382004, "frac_reward_zero_std": 0.5, "grad_norm": 0.5929316878318787, "learning_rate": 1e-06, "loss": 0.0269, "num_tokens": 746574640.0, "reward": 0.484375, "reward_std": 0.2098345011472702, "rewards/simpleverify_reward/mean": 0.484375, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1902, "tools/generated_tokens": 3636.88671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2008.0, "completions/max_terminated_length": 2008.0, "completions/mean_length": 914.6953125, "completions/mean_terminated_length": 914.6953125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.093362245708704, "epoch": 0.32428057170852237, "frac_reward_zero_std": 0.5, "grad_norm": 0.7593372464179993, "learning_rate": 1e-06, "loss": -0.0002, "num_tokens": 746890178.0, "reward": 0.46484375, "reward_std": 0.17906175553798676, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1903, "tools/generated_tokens": 3706.69140625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2044.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 942.57421875, "completions/mean_terminated_length": 942.57421875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.08218329679220915, "epoch": 0.3244509766332247, "frac_reward_zero_std": 0.5, "grad_norm": 0.6194080710411072, "learning_rate": 1e-06, "loss": -0.0419, "num_tokens": 747209893.0, "reward": 0.39453125, "reward_std": 0.22989481687545776, "rewards/simpleverify_reward/mean": 0.39453125, "rewards/simpleverify_reward/std": 0.48970720171928406, "step": 1904, "tools/generated_tokens": 4214.58984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.59765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2043.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 971.9765625, "completions/mean_terminated_length": 971.9765625, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.09346222272142768, "epoch": 0.324621381557927, "frac_reward_zero_std": 0.5, "grad_norm": 0.4486996829509735, "learning_rate": 1e-06, "loss": 0.0175, "num_tokens": 747530703.0, "reward": 0.61328125, "reward_std": 0.20656242966651917, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1905, "tools/generated_tokens": 3067.984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2034.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1088.5703125, "completions/mean_terminated_length": 1088.5703125, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.09619740257039666, "epoch": 0.32479178648262935, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6381232738494873, "learning_rate": 1e-06, "loss": -0.0607, "num_tokens": 747878401.0, "reward": 0.5390625, "reward_std": 0.20015865564346313, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1906, "tools/generated_tokens": 3216.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2024.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 929.61328125, "completions/mean_terminated_length": 929.61328125, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.09160624677315354, "epoch": 0.3249621914073317, "frac_reward_zero_std": 0.625, "grad_norm": 0.4725303649902344, "learning_rate": 1e-06, "loss": -0.0047, "num_tokens": 748189758.0, "reward": 0.734375, "reward_std": 0.1454564929008484, "rewards/simpleverify_reward/mean": 0.734375, "rewards/simpleverify_reward/std": 0.4425306022167206, "step": 1907, "tools/generated_tokens": 3025.62109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.0234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 885.4609375, "completions/mean_terminated_length": 876.3070678710938, "completions/min_length": 7.0, "completions/min_terminated_length": 7.0, "entropy": 0.0997252780944109, "epoch": 0.325132596332034, "frac_reward_zero_std": 0.375, "grad_norm": 0.9593226313591003, "learning_rate": 1e-06, "loss": -0.0109, "num_tokens": 748515732.0, "reward": 0.515625, "reward_std": 0.24205546081066132, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1908, "tools/generated_tokens": 3845.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1021.3828125, "completions/mean_terminated_length": 1017.35693359375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.08760990481823683, "epoch": 0.32530300125673633, "frac_reward_zero_std": 0.375, "grad_norm": 0.6139023900032043, "learning_rate": 1e-06, "loss": 0.0094, "num_tokens": 748855014.0, "reward": 0.5390625, "reward_std": 0.27564120292663574, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1909, "tools/generated_tokens": 3973.38671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.44140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2022.0, "completions/max_terminated_length": 2022.0, "completions/mean_length": 1009.4296875, "completions/mean_terminated_length": 1009.4296875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.07973799714818597, "epoch": 0.32547340618143866, "frac_reward_zero_std": 0.5625, "grad_norm": 0.5801877975463867, "learning_rate": 1e-06, "loss": -0.0089, "num_tokens": 749180612.0, "reward": 0.640625, "reward_std": 0.14853152632713318, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1910, "tools/generated_tokens": 3145.421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2023.0, "completions/mean_length": 1084.78515625, "completions/mean_terminated_length": 1081.0079345703125, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 0.08412523521110415, "epoch": 0.325643811106141, "frac_reward_zero_std": 0.5, "grad_norm": 0.5232059955596924, "learning_rate": 1e-06, "loss": -0.042, "num_tokens": 749527949.0, "reward": 0.51953125, "reward_std": 0.18640750646591187, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1911, "tools/generated_tokens": 3524.78515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.19140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2045.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 873.32421875, "completions/mean_terminated_length": 873.32421875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.08658592775464058, "epoch": 0.3258142160308433, "frac_reward_zero_std": 0.375, "grad_norm": 0.514909029006958, "learning_rate": 1e-06, "loss": 0.0014, "num_tokens": 749834288.0, "reward": 0.296875, "reward_std": 0.23528218269348145, "rewards/simpleverify_reward/mean": 0.296875, "rewards/simpleverify_reward/std": 0.45777595043182373, "step": 1912, "tools/generated_tokens": 4233.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 953.09765625, "completions/mean_terminated_length": 948.803955078125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.0756278308108449, "epoch": 0.3259846209555456, "frac_reward_zero_std": 0.375, "grad_norm": 0.6880056262016296, "learning_rate": 1e-06, "loss": -0.0537, "num_tokens": 750154841.0, "reward": 0.5859375, "reward_std": 0.22742824256420135, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1913, "tools/generated_tokens": 3545.09765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2042.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1063.1015625, "completions/mean_terminated_length": 1063.1015625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0895443968474865, "epoch": 0.3261550258802479, "frac_reward_zero_std": 0.25, "grad_norm": 0.5069884657859802, "learning_rate": 1e-06, "loss": -0.0114, "num_tokens": 750501363.0, "reward": 0.6171875, "reward_std": 0.2593595087528229, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1914, "tools/generated_tokens": 3783.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 968.83984375, "completions/mean_terminated_length": 964.60791015625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.0772891603410244, "epoch": 0.32632543080495025, "frac_reward_zero_std": 0.5625, "grad_norm": 0.39360857009887695, "learning_rate": 1e-06, "loss": -0.0232, "num_tokens": 750820378.0, "reward": 0.5859375, "reward_std": 0.1928790807723999, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1915, "tools/generated_tokens": 3264.85546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.12109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 985.46484375, "completions/mean_terminated_length": 985.46484375, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.07483136327937245, "epoch": 0.3264958357296526, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4292159080505371, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 751151073.0, "reward": 0.54296875, "reward_std": 0.2334996908903122, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1916, "tools/generated_tokens": 4329.46484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1053.09375, "completions/mean_terminated_length": 1049.1922607421875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.09218379901722074, "epoch": 0.3266662406543549, "frac_reward_zero_std": 0.5625, "grad_norm": 0.39092278480529785, "learning_rate": 1e-06, "loss": -0.0042, "num_tokens": 751508969.0, "reward": 0.4765625, "reward_std": 0.15734326839447021, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1917, "tools/generated_tokens": 3981.109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1147.19921875, "completions/mean_terminated_length": 1118.14111328125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.09617560729384422, "epoch": 0.32683664557905723, "frac_reward_zero_std": 0.5625, "grad_norm": 1.4030731916427612, "learning_rate": 1e-06, "loss": -0.0148, "num_tokens": 751885164.0, "reward": 0.52734375, "reward_std": 0.1630660742521286, "rewards/simpleverify_reward/mean": 0.52734375, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1918, "tools/generated_tokens": 4491.20703125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1183.0703125, "completions/mean_terminated_length": 1169.34130859375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.10042772768065333, "epoch": 0.32700705050375956, "frac_reward_zero_std": 0.5625, "grad_norm": 0.38762757182121277, "learning_rate": 1e-06, "loss": 0.0029, "num_tokens": 752253230.0, "reward": 0.7265625, "reward_std": 0.16593991219997406, "rewards/simpleverify_reward/mean": 0.7265625, "rewards/simpleverify_reward/std": 0.446596622467041, "step": 1919, "tools/generated_tokens": 3351.07421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1109.1796875, "completions/mean_terminated_length": 1105.498046875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.0955316056497395, "epoch": 0.3271774554284619, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4087553918361664, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 752613068.0, "reward": 0.5, "reward_std": 0.26221945881843567, "rewards/simpleverify_reward/mean": 0.5, "rewards/simpleverify_reward/std": 0.5009794235229492, "step": 1920, "tools/generated_tokens": 3613.19921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.22265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1126.44921875, "completions/mean_terminated_length": 1122.8353271484375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.12396797770634294, "epoch": 0.3273478603531642, "frac_reward_zero_std": 0.5, "grad_norm": 0.45202717185020447, "learning_rate": 1e-06, "loss": -0.0067, "num_tokens": 752970351.0, "reward": 0.63671875, "reward_std": 0.2035529911518097, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1921, "tools/generated_tokens": 3366.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.09375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1101.5859375, "completions/mean_terminated_length": 1097.8746337890625, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.12630308791995049, "epoch": 0.32751826527786654, "frac_reward_zero_std": 0.25, "grad_norm": 0.4518052041530609, "learning_rate": 1e-06, "loss": -0.0304, "num_tokens": 753324421.0, "reward": 0.63671875, "reward_std": 0.28552472591400146, "rewards/simpleverify_reward/mean": 0.63671875, "rewards/simpleverify_reward/std": 0.48188701272010803, "step": 1922, "tools/generated_tokens": 3597.58203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1057.50390625, "completions/mean_terminated_length": 1041.7818603515625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.1151580954901874, "epoch": 0.32768867020256887, "frac_reward_zero_std": 0.25, "grad_norm": 0.4318159520626068, "learning_rate": 1e-06, "loss": 0.0248, "num_tokens": 753682294.0, "reward": 0.5390625, "reward_std": 0.2903963327407837, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1923, "tools/generated_tokens": 4505.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.68359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1064.5859375, "completions/mean_terminated_length": 1060.7294921875, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 0.13212102372199297, "epoch": 0.3278590751272712, "frac_reward_zero_std": 0.375, "grad_norm": 0.4426436722278595, "learning_rate": 1e-06, "loss": 0.0161, "num_tokens": 754032428.0, "reward": 0.69921875, "reward_std": 0.2427206039428711, "rewards/simpleverify_reward/mean": 0.69921875, "rewards/simpleverify_reward/std": 0.45949608087539673, "step": 1924, "tools/generated_tokens": 3736.5859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3046875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1133.12890625, "completions/mean_terminated_length": 1125.9251708984375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.12676287768408656, "epoch": 0.3280294800519735, "frac_reward_zero_std": 0.4375, "grad_norm": 0.3979257047176361, "learning_rate": 1e-06, "loss": -0.0004, "num_tokens": 754395629.0, "reward": 0.74609375, "reward_std": 0.18948253989219666, "rewards/simpleverify_reward/mean": 0.74609375, "rewards/simpleverify_reward/std": 0.4360972046852112, "step": 1925, "tools/generated_tokens": 3517.1328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2034.0, "completions/mean_length": 1124.38671875, "completions/mean_terminated_length": 1117.1141357421875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.11855522030964494, "epoch": 0.32819988497667585, "frac_reward_zero_std": 0.625, "grad_norm": 0.397084504365921, "learning_rate": 1e-06, "loss": 0.0245, "num_tokens": 754754272.0, "reward": 0.625, "reward_std": 0.1544942855834961, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1926, "tools/generated_tokens": 3260.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1075.3125, "completions/mean_terminated_length": 1063.7786865234375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.13150327745825052, "epoch": 0.3283702899013782, "frac_reward_zero_std": 0.625, "grad_norm": 0.35653120279312134, "learning_rate": 1e-06, "loss": 0.004, "num_tokens": 755093632.0, "reward": 0.5234375, "reward_std": 0.15701062977313995, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1927, "tools/generated_tokens": 3571.3203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.21875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1120.671875, "completions/mean_terminated_length": 1105.952392578125, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1392094404436648, "epoch": 0.32854069482608045, "frac_reward_zero_std": 0.5, "grad_norm": 0.4398752748966217, "learning_rate": 1e-06, "loss": 0.0031, "num_tokens": 755454540.0, "reward": 0.3125, "reward_std": 0.2123890072107315, "rewards/simpleverify_reward/mean": 0.3125, "rewards/simpleverify_reward/std": 0.4644203782081604, "step": 1928, "tools/generated_tokens": 3968.66796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1100.55078125, "completions/mean_terminated_length": 1089.3162841796875, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.14392456971108913, "epoch": 0.3287110997507828, "frac_reward_zero_std": 0.1875, "grad_norm": 0.34651613235473633, "learning_rate": 1e-06, "loss": -0.0118, "num_tokens": 755815017.0, "reward": 0.61328125, "reward_std": 0.2674104869365692, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1929, "tools/generated_tokens": 3508.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.17578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1116.640625, "completions/mean_terminated_length": 1112.98828125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.12356674671173096, "epoch": 0.3288815046754851, "frac_reward_zero_std": 0.625, "grad_norm": 0.3066537082195282, "learning_rate": 1e-06, "loss": 0.0007, "num_tokens": 756159373.0, "reward": 0.7890625, "reward_std": 0.13566282391548157, "rewards/simpleverify_reward/mean": 0.7890625, "rewards/simpleverify_reward/std": 0.4087733030319214, "step": 1930, "tools/generated_tokens": 3100.640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.96875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1101.75, "completions/mean_terminated_length": 1098.039306640625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.1128494655713439, "epoch": 0.32905190960018743, "frac_reward_zero_std": 0.625, "grad_norm": 0.27842676639556885, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 756515133.0, "reward": 0.546875, "reward_std": 0.14516396820545197, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1931, "tools/generated_tokens": 3429.765625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.13671875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1185.109375, "completions/mean_terminated_length": 1164.4000244140625, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.1133117345161736, "epoch": 0.32922231452488976, "frac_reward_zero_std": 0.375, "grad_norm": 0.38291695713996887, "learning_rate": 1e-06, "loss": -0.0011, "num_tokens": 756897065.0, "reward": 0.47265625, "reward_std": 0.26286041736602783, "rewards/simpleverify_reward/mean": 0.47265625, "rewards/simpleverify_reward/std": 0.5002297759056091, "step": 1932, "tools/generated_tokens": 4417.12109375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1103.10546875, "completions/mean_terminated_length": 1072.625, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.11755352793261409, "epoch": 0.3293927194495921, "frac_reward_zero_std": 0.4375, "grad_norm": 0.45664823055267334, "learning_rate": 1e-06, "loss": -0.0182, "num_tokens": 757255060.0, "reward": 0.53515625, "reward_std": 0.18276195228099823, "rewards/simpleverify_reward/mean": 0.53515625, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1933, "tools/generated_tokens": 4047.10546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1005.62890625, "completions/mean_terminated_length": 997.4212646484375, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.1282304567284882, "epoch": 0.3295631243742944, "frac_reward_zero_std": 0.6875, "grad_norm": 0.4013802111148834, "learning_rate": 1e-06, "loss": -0.0055, "num_tokens": 757585221.0, "reward": 0.46484375, "reward_std": 0.13926705718040466, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1934, "tools/generated_tokens": 4421.63671875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.66796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1121.36328125, "completions/mean_terminated_length": 1117.7294921875, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.10692885098978877, "epoch": 0.32973352929899674, "frac_reward_zero_std": 0.625, "grad_norm": 0.3182205855846405, "learning_rate": 1e-06, "loss": -0.009, "num_tokens": 757931762.0, "reward": 0.515625, "reward_std": 0.14271603524684906, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1935, "tools/generated_tokens": 2849.36328125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.84375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2030.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1111.57421875, "completions/mean_terminated_length": 1111.57421875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.13573657860979438, "epoch": 0.32990393422369907, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4673006236553192, "learning_rate": 1e-06, "loss": 0.0016, "num_tokens": 758292453.0, "reward": 0.6796875, "reward_std": 0.2759421765804291, "rewards/simpleverify_reward/mean": 0.6796875, "rewards/simpleverify_reward/std": 0.4675106406211853, "step": 1936, "tools/generated_tokens": 3543.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1143.89453125, "completions/mean_terminated_length": 1122.196044921875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.11510316748172045, "epoch": 0.3300743391484014, "frac_reward_zero_std": 0.375, "grad_norm": 0.3383469581604004, "learning_rate": 1e-06, "loss": -0.0079, "num_tokens": 758669482.0, "reward": 0.5859375, "reward_std": 0.2191799283027649, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1937, "tools/generated_tokens": 3719.89453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1100.046875, "completions/mean_terminated_length": 1057.4857177734375, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.10329854488372803, "epoch": 0.3302447440731037, "frac_reward_zero_std": 0.4375, "grad_norm": 0.9953184127807617, "learning_rate": 1e-06, "loss": 0.0497, "num_tokens": 759022134.0, "reward": 0.5703125, "reward_std": 0.22645533084869385, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1938, "tools/generated_tokens": 3996.046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2036.0, "completions/mean_length": 1166.12109375, "completions/mean_terminated_length": 1137.67333984375, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.13397061033174396, "epoch": 0.33041514899780605, "frac_reward_zero_std": 0.6875, "grad_norm": 0.40637320280075073, "learning_rate": 1e-06, "loss": 0.0082, "num_tokens": 759404549.0, "reward": 0.46875, "reward_std": 0.13661926984786987, "rewards/simpleverify_reward/mean": 0.46875, "rewards/simpleverify_reward/std": 0.5, "step": 1939, "tools/generated_tokens": 4062.1171875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4140625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2024.0, "completions/mean_length": 1138.75390625, "completions/mean_terminated_length": 1120.6414794921875, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.10119526507332921, "epoch": 0.3305855539225084, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4420066177845001, "learning_rate": 1e-06, "loss": 0.0185, "num_tokens": 759781110.0, "reward": 0.4921875, "reward_std": 0.25207996368408203, "rewards/simpleverify_reward/mean": 0.4921875, "rewards/simpleverify_reward/std": 0.5009182691574097, "step": 1940, "tools/generated_tokens": 4338.78125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2040.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1120.1953125, "completions/mean_terminated_length": 1120.1953125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.09602189529687166, "epoch": 0.3307559588472107, "frac_reward_zero_std": 0.4375, "grad_norm": 0.40170466899871826, "learning_rate": 1e-06, "loss": -0.0006, "num_tokens": 760142472.0, "reward": 0.609375, "reward_std": 0.24210743606090546, "rewards/simpleverify_reward/mean": 0.609375, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1941, "tools/generated_tokens": 3568.203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1058.234375, "completions/mean_terminated_length": 1054.35302734375, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.07727785594761372, "epoch": 0.33092636377191303, "frac_reward_zero_std": 0.625, "grad_norm": 0.5055992603302002, "learning_rate": 1e-06, "loss": -0.0147, "num_tokens": 760484740.0, "reward": 0.66015625, "reward_std": 0.1318160742521286, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1942, "tools/generated_tokens": 2914.23046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.90625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1135.83984375, "completions/mean_terminated_length": 1091.036865234375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.08401553332805634, "epoch": 0.3310967686966153, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3892535865306854, "learning_rate": 1e-06, "loss": 0.0086, "num_tokens": 760849803.0, "reward": 0.44140625, "reward_std": 0.17399311065673828, "rewards/simpleverify_reward/mean": 0.44140625, "rewards/simpleverify_reward/std": 0.4975275993347168, "step": 1943, "tools/generated_tokens": 4639.8984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.7109375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2037.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1032.68359375, "completions/mean_terminated_length": 1032.68359375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.10222691390663385, "epoch": 0.33126717362131763, "frac_reward_zero_std": 0.375, "grad_norm": 0.4247925579547882, "learning_rate": 1e-06, "loss": -0.0009, "num_tokens": 761192106.0, "reward": 0.6484375, "reward_std": 0.23448145389556885, "rewards/simpleverify_reward/mean": 0.6484375, "rewards/simpleverify_reward/std": 0.47839346528053284, "step": 1944, "tools/generated_tokens": 3216.6796875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.06640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1124.953125, "completions/mean_terminated_length": 1117.68505859375, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.08855460397899151, "epoch": 0.33143757854601996, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4909963607788086, "learning_rate": 1e-06, "loss": 0.0105, "num_tokens": 761553502.0, "reward": 0.4765625, "reward_std": 0.1978301852941513, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1945, "tools/generated_tokens": 3732.94921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2734375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1141.84765625, "completions/mean_terminated_length": 1131.102783203125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.09569338383153081, "epoch": 0.3316079834707223, "frac_reward_zero_std": 0.3125, "grad_norm": 0.43997520208358765, "learning_rate": 1e-06, "loss": 0.013, "num_tokens": 761916999.0, "reward": 0.58203125, "reward_std": 0.252851665019989, "rewards/simpleverify_reward/mean": 0.58203125, "rewards/simpleverify_reward/std": 0.49419113993644714, "step": 1946, "tools/generated_tokens": 3133.8515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.97265625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1068.2890625, "completions/mean_terminated_length": 1060.5748291015625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.0820078831166029, "epoch": 0.3317783883954246, "frac_reward_zero_std": 0.5, "grad_norm": 0.45428481698036194, "learning_rate": 1e-06, "loss": 0.0005, "num_tokens": 762260177.0, "reward": 0.640625, "reward_std": 0.19519630074501038, "rewards/simpleverify_reward/mean": 0.640625, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1947, "tools/generated_tokens": 3708.2734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.2890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1140.58203125, "completions/mean_terminated_length": 1133.43701171875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.08309565996751189, "epoch": 0.33194879332012694, "frac_reward_zero_std": 0.625, "grad_norm": 0.33417990803718567, "learning_rate": 1e-06, "loss": -0.015, "num_tokens": 762617126.0, "reward": 0.4296875, "reward_std": 0.12621080875396729, "rewards/simpleverify_reward/mean": 0.4296875, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1948, "tools/generated_tokens": 3308.46875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.05859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1152.69921875, "completions/mean_terminated_length": 1120.076904296875, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.07624175865203142, "epoch": 0.3321191982448293, "frac_reward_zero_std": 0.5625, "grad_norm": 0.4717263877391815, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 762979897.0, "reward": 0.6171875, "reward_std": 0.14886415004730225, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1949, "tools/generated_tokens": 3576.6953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1186.5546875, "completions/mean_terminated_length": 1165.884033203125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.07730635534971952, "epoch": 0.3322896031695316, "frac_reward_zero_std": 0.3125, "grad_norm": 0.5689806342124939, "learning_rate": 1e-06, "loss": 0.0314, "num_tokens": 763363735.0, "reward": 0.54296875, "reward_std": 0.291836142539978, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1950, "tools/generated_tokens": 3970.56640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1187.7421875, "completions/mean_terminated_length": 1114.8389892578125, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.07509825192391872, "epoch": 0.3324600080942339, "frac_reward_zero_std": 0.625, "grad_norm": 0.3713930547237396, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 763739093.0, "reward": 0.359375, "reward_std": 0.1636136770248413, "rewards/simpleverify_reward/mean": 0.359375, "rewards/simpleverify_reward/std": 0.4807571768760681, "step": 1951, "tools/generated_tokens": 4379.73828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.55859375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1081.35546875, "completions/mean_terminated_length": 1077.5648193359375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.07854221481829882, "epoch": 0.33263041301893626, "frac_reward_zero_std": 0.3125, "grad_norm": 0.6061356067657471, "learning_rate": 1e-06, "loss": 0.0159, "num_tokens": 764087440.0, "reward": 0.5859375, "reward_std": 0.2754891514778137, "rewards/simpleverify_reward/mean": 0.5859375, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1952, "tools/generated_tokens": 3217.3515625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.04296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1196.98828125, "completions/mean_terminated_length": 1180.035888671875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.08526395447552204, "epoch": 0.3328008179436386, "frac_reward_zero_std": 0.625, "grad_norm": 0.319280743598938, "learning_rate": 1e-06, "loss": 0.0007, "num_tokens": 764465549.0, "reward": 0.5625, "reward_std": 0.1312704086303711, "rewards/simpleverify_reward/mean": 0.5625, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1953, "tools/generated_tokens": 3668.9921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.20703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2046.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1101.421875, "completions/mean_terminated_length": 1101.421875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.07984186476096511, "epoch": 0.3329712228683409, "frac_reward_zero_std": 0.625, "grad_norm": 0.30380526185035706, "learning_rate": 1e-06, "loss": -0.0097, "num_tokens": 764800521.0, "reward": 0.5703125, "reward_std": 0.11509781330823898, "rewards/simpleverify_reward/mean": 0.5703125, "rewards/simpleverify_reward/std": 0.4960011839866638, "step": 1954, "tools/generated_tokens": 2677.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.76953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1147.9296875, "completions/mean_terminated_length": 1140.842529296875, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.06732810591347516, "epoch": 0.33314162779304324, "frac_reward_zero_std": 0.5625, "grad_norm": 0.33665722608566284, "learning_rate": 1e-06, "loss": -0.0103, "num_tokens": 765154599.0, "reward": 0.5546875, "reward_std": 0.14635254442691803, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1955, "tools/generated_tokens": 3235.9296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.01953125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2038.0, "completions/mean_length": 1178.1015625, "completions/mean_terminated_length": 1135.319580078125, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.07550344476476312, "epoch": 0.33331203271774557, "frac_reward_zero_std": 0.5625, "grad_norm": 0.42035120725631714, "learning_rate": 1e-06, "loss": 0.0146, "num_tokens": 765530609.0, "reward": 0.2890625, "reward_std": 0.19918768107891083, "rewards/simpleverify_reward/mean": 0.2890625, "rewards/simpleverify_reward/std": 0.45421501994132996, "step": 1956, "tools/generated_tokens": 4482.125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.61328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1162.29296875, "completions/mean_terminated_length": 1158.8197021484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.08386320946738124, "epoch": 0.3334824376424479, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4419395923614502, "learning_rate": 1e-06, "loss": -0.0271, "num_tokens": 765897660.0, "reward": 0.4140625, "reward_std": 0.16581955552101135, "rewards/simpleverify_reward/mean": 0.4140625, "rewards/simpleverify_reward/std": 0.4935242533683777, "step": 1957, "tools/generated_tokens": 3810.3125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.29296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1170.828125, "completions/mean_terminated_length": 1163.9212646484375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.08224812569096684, "epoch": 0.33365284256715017, "frac_reward_zero_std": 0.5, "grad_norm": 0.4407566785812378, "learning_rate": 1e-06, "loss": -0.0118, "num_tokens": 766261520.0, "reward": 0.7578125, "reward_std": 0.21564999222755432, "rewards/simpleverify_reward/mean": 0.7578125, "rewards/simpleverify_reward/std": 0.4292463958263397, "step": 1958, "tools/generated_tokens": 2810.83203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.80078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1115.515625, "completions/mean_terminated_length": 1108.1732177734375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.07163693197071552, "epoch": 0.3338232474918525, "frac_reward_zero_std": 0.3125, "grad_norm": 0.740218997001648, "learning_rate": 1e-06, "loss": 0.0226, "num_tokens": 766632228.0, "reward": 0.453125, "reward_std": 0.28663188219070435, "rewards/simpleverify_reward/mean": 0.453125, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1959, "tools/generated_tokens": 4163.5234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.48828125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1189.19140625, "completions/mean_terminated_length": 1182.4290771484375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.0809728167951107, "epoch": 0.3339936524165548, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3779401481151581, "learning_rate": 1e-06, "loss": 0.0195, "num_tokens": 766996965.0, "reward": 0.72265625, "reward_std": 0.16926807165145874, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1960, "tools/generated_tokens": 3189.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.9765625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1164.71484375, "completions/mean_terminated_length": 1143.51611328125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.09033261798322201, "epoch": 0.33416405734125715, "frac_reward_zero_std": 0.6875, "grad_norm": 0.38102635741233826, "learning_rate": 1e-06, "loss": -0.0182, "num_tokens": 767378268.0, "reward": 0.578125, "reward_std": 0.1161910742521286, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1961, "tools/generated_tokens": 3884.71875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2039.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1122.38671875, "completions/mean_terminated_length": 1122.38671875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.09322395268827677, "epoch": 0.3343344622659595, "frac_reward_zero_std": 0.6875, "grad_norm": 0.3577176630496979, "learning_rate": 1e-06, "loss": 0.0098, "num_tokens": 767747711.0, "reward": 0.50390625, "reward_std": 0.078125, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1962, "tools/generated_tokens": 3978.390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.39453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1117.0078125, "completions/mean_terminated_length": 1113.35693359375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.07700205058790743, "epoch": 0.3345048671906618, "frac_reward_zero_std": 0.3125, "grad_norm": 0.6411166191101074, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 768110673.0, "reward": 0.55078125, "reward_std": 0.2583726942539215, "rewards/simpleverify_reward/mean": 0.55078125, "rewards/simpleverify_reward/std": 0.49838894605636597, "step": 1963, "tools/generated_tokens": 4005.015625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1174.21875, "completions/mean_terminated_length": 1160.3492431640625, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.07681906968355179, "epoch": 0.33467527211536413, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5305586457252502, "learning_rate": 1e-06, "loss": -0.0117, "num_tokens": 768495945.0, "reward": 0.515625, "reward_std": 0.23414883017539978, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1964, "tools/generated_tokens": 4262.22265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1216.37109375, "completions/mean_terminated_length": 1206.513916015625, "completions/min_length": 1.0, "completions/min_terminated_length": 1.0, "entropy": 0.07484760507941246, "epoch": 0.33484567704006646, "frac_reward_zero_std": 0.5625, "grad_norm": 0.32089054584503174, "learning_rate": 1e-06, "loss": -0.0169, "num_tokens": 768868728.0, "reward": 0.6328125, "reward_std": 0.16890643537044525, "rewards/simpleverify_reward/mean": 0.6328125, "rewards/simpleverify_reward/std": 0.48298248648643494, "step": 1965, "tools/generated_tokens": 2760.3828125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.75390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1158.61328125, "completions/mean_terminated_length": 1151.6102294921875, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.08253148477524519, "epoch": 0.3350160819647688, "frac_reward_zero_std": 0.375, "grad_norm": 0.49426499009132385, "learning_rate": 1e-06, "loss": -0.0011, "num_tokens": 769246389.0, "reward": 0.56640625, "reward_std": 0.25340840220451355, "rewards/simpleverify_reward/mean": 0.56640625, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1966, "tools/generated_tokens": 4062.609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.41796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1115.3984375, "completions/mean_terminated_length": 1081.4169921875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.0754284868016839, "epoch": 0.3351864868894711, "frac_reward_zero_std": 0.375, "grad_norm": 0.5146631598472595, "learning_rate": 1e-06, "loss": 0.0327, "num_tokens": 769612747.0, "reward": 0.4765625, "reward_std": 0.2657455503940582, "rewards/simpleverify_reward/mean": 0.4765625, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1967, "tools/generated_tokens": 4339.40625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.57421875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2035.0, "completions/mean_length": 1143.62890625, "completions/mean_terminated_length": 1114.4595947265625, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.08422506880015135, "epoch": 0.33535689181417344, "frac_reward_zero_std": 0.6875, "grad_norm": 0.29242125153541565, "learning_rate": 1e-06, "loss": 0.0137, "num_tokens": 769987852.0, "reward": 0.66015625, "reward_std": 0.1372220814228058, "rewards/simpleverify_reward/mean": 0.66015625, "rewards/simpleverify_reward/std": 0.47458380460739136, "step": 1968, "tools/generated_tokens": 3847.62890625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1171.19140625, "completions/mean_terminated_length": 1139.242919921875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.07960205734707415, "epoch": 0.33552729673887577, "frac_reward_zero_std": 0.5, "grad_norm": 0.34981444478034973, "learning_rate": 1e-06, "loss": 0.0411, "num_tokens": 770351357.0, "reward": 0.671875, "reward_std": 0.20244863629341125, "rewards/simpleverify_reward/mean": 0.671875, "rewards/simpleverify_reward/std": 0.47045037150382996, "step": 1969, "tools/generated_tokens": 3115.1875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.94921875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2042.0, "completions/mean_length": 1198.29296875, "completions/mean_terminated_length": 1191.6024169921875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0930467490106821, "epoch": 0.3356977016635781, "frac_reward_zero_std": 0.1875, "grad_norm": 0.5678503513336182, "learning_rate": 1e-06, "loss": 0.0092, "num_tokens": 770731720.0, "reward": 0.578125, "reward_std": 0.317813515663147, "rewards/simpleverify_reward/mean": 0.578125, "rewards/simpleverify_reward/std": 0.49482619762420654, "step": 1970, "tools/generated_tokens": 3558.29296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1141.83984375, "completions/mean_terminated_length": 1123.788818359375, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.08626068150624633, "epoch": 0.3358681065882804, "frac_reward_zero_std": 0.5, "grad_norm": 0.3982565104961395, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 771099695.0, "reward": 0.31640625, "reward_std": 0.17917026579380035, "rewards/simpleverify_reward/mean": 0.31640625, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1971, "tools/generated_tokens": 4125.84375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.45703125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1130.4453125, "completions/mean_terminated_length": 1119.5653076171875, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.0945667251944542, "epoch": 0.33603851151298275, "frac_reward_zero_std": 0.4375, "grad_norm": 0.5123549103736877, "learning_rate": 1e-06, "loss": 0.0074, "num_tokens": 771467105.0, "reward": 0.46484375, "reward_std": 0.22005823254585266, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1972, "tools/generated_tokens": 3954.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.37890625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1200.390625, "completions/mean_terminated_length": 1186.9405517578125, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.08354484708979726, "epoch": 0.336208916437685, "frac_reward_zero_std": 0.25, "grad_norm": 0.4878859519958496, "learning_rate": 1e-06, "loss": 0.0084, "num_tokens": 771842053.0, "reward": 0.625, "reward_std": 0.2923789918422699, "rewards/simpleverify_reward/mean": 0.625, "rewards/simpleverify_reward/std": 0.4850712716579437, "step": 1973, "tools/generated_tokens": 3632.40234375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1232.0078125, "completions/mean_terminated_length": 1225.5826416015625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.0820220010355115, "epoch": 0.33637932136238735, "frac_reward_zero_std": 0.5, "grad_norm": 0.3647536635398865, "learning_rate": 1e-06, "loss": 0.0026, "num_tokens": 772219927.0, "reward": 0.48046875, "reward_std": 0.15646304190158844, "rewards/simpleverify_reward/mean": 0.48046875, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1974, "tools/generated_tokens": 3352.00390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.03515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1137.234375, "completions/mean_terminated_length": 1126.438720703125, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.08003668813034892, "epoch": 0.3365497262870897, "frac_reward_zero_std": 0.5625, "grad_norm": 0.39166730642318726, "learning_rate": 1e-06, "loss": -0.0142, "num_tokens": 772591459.0, "reward": 0.70703125, "reward_std": 0.1679086685180664, "rewards/simpleverify_reward/mean": 0.70703125, "rewards/simpleverify_reward/std": 0.45601576566696167, "step": 1975, "tools/generated_tokens": 3793.2578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1283.5703125, "completions/mean_terminated_length": 1277.5511474609375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.08399980142712593, "epoch": 0.336720131211792, "frac_reward_zero_std": 0.375, "grad_norm": 0.35710886120796204, "learning_rate": 1e-06, "loss": 0.0066, "num_tokens": 772985717.0, "reward": 0.51953125, "reward_std": 0.20148904621601105, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1976, "tools/generated_tokens": 3179.57421875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.92578125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2041.0, "completions/mean_length": 1257.515625, "completions/mean_terminated_length": 1228.7125244140625, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.06657208013348281, "epoch": 0.33689053613649433, "frac_reward_zero_std": 0.625, "grad_norm": 0.23854054510593414, "learning_rate": 1e-06, "loss": -0.0117, "num_tokens": 773374393.0, "reward": 0.68359375, "reward_std": 0.13614009320735931, "rewards/simpleverify_reward/mean": 0.68359375, "rewards/simpleverify_reward/std": 0.4659844934940338, "step": 1977, "tools/generated_tokens": 3545.52734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1178.44921875, "completions/mean_terminated_length": 1175.039306640625, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.09179374109953642, "epoch": 0.33706094106119666, "frac_reward_zero_std": 0.625, "grad_norm": 0.32337436079978943, "learning_rate": 1e-06, "loss": -0.0043, "num_tokens": 773755628.0, "reward": 0.46484375, "reward_std": 0.13534127175807953, "rewards/simpleverify_reward/mean": 0.46484375, "rewards/simpleverify_reward/std": 0.49973952770233154, "step": 1978, "tools/generated_tokens": 3874.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.31640625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1229.8515625, "completions/mean_terminated_length": 1223.409423828125, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.10396990459412336, "epoch": 0.337231345985899, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4773634374141693, "learning_rate": 1e-06, "loss": -0.0012, "num_tokens": 774142678.0, "reward": 0.5390625, "reward_std": 0.2320467084646225, "rewards/simpleverify_reward/mean": 0.5390625, "rewards/simpleverify_reward/std": 0.4994482398033142, "step": 1979, "tools/generated_tokens": 3933.859375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3203125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2040.0, "completions/mean_length": 1143.46875, "completions/mean_terminated_length": 1136.346435546875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.07877062773332, "epoch": 0.3374017509106013, "frac_reward_zero_std": 0.25, "grad_norm": 0.5208812952041626, "learning_rate": 1e-06, "loss": -0.0032, "num_tokens": 774511726.0, "reward": 0.51953125, "reward_std": 0.31239640712738037, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1980, "tools/generated_tokens": 4263.4609375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1165.32421875, "completions/mean_terminated_length": 1161.86279296875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.0853398465551436, "epoch": 0.33757215583530364, "frac_reward_zero_std": 0.25, "grad_norm": 0.4581199884414673, "learning_rate": 1e-06, "loss": -0.0113, "num_tokens": 774893585.0, "reward": 0.50390625, "reward_std": 0.26275384426116943, "rewards/simpleverify_reward/mean": 0.50390625, "rewards/simpleverify_reward/std": 0.5009641647338867, "step": 1981, "tools/generated_tokens": 4093.33984375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.4296875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2037.0, "completions/mean_length": 1179.9375, "completions/mean_terminated_length": 1176.533447265625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.076184606179595, "epoch": 0.33774256076000597, "frac_reward_zero_std": 0.3125, "grad_norm": 0.4168841242790222, "learning_rate": 1e-06, "loss": 0.0289, "num_tokens": 775270657.0, "reward": 0.58984375, "reward_std": 0.2921258509159088, "rewards/simpleverify_reward/mean": 0.58984375, "rewards/simpleverify_reward/std": 0.49282538890838623, "step": 1982, "tools/generated_tokens": 4043.9375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1256.4296875, "completions/mean_terminated_length": 1256.4296875, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.07739983219653368, "epoch": 0.3379129656847083, "frac_reward_zero_std": 0.5625, "grad_norm": 0.29777535796165466, "learning_rate": 1e-06, "loss": 0.0125, "num_tokens": 775662559.0, "reward": 0.67578125, "reward_std": 0.12082062661647797, "rewards/simpleverify_reward/mean": 0.67578125, "rewards/simpleverify_reward/std": 0.46899911761283875, "step": 1983, "tools/generated_tokens": 3120.42578125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.91015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2044.0, "completions/mean_length": 1234.19140625, "completions/mean_terminated_length": 1221.27392578125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.09374697180464864, "epoch": 0.3380833706094106, "frac_reward_zero_std": 0.5, "grad_norm": 0.4631107747554779, "learning_rate": 1e-06, "loss": 0.0097, "num_tokens": 776053360.0, "reward": 0.43359375, "reward_std": 0.18914085626602173, "rewards/simpleverify_reward/mean": 0.43359375, "rewards/simpleverify_reward/std": 0.4965413510799408, "step": 1984, "tools/generated_tokens": 4298.1953125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.49609375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2027.0, "completions/mean_length": 1177.0, "completions/mean_terminated_length": 1166.6719970703125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.07807535585016012, "epoch": 0.33825377553411295, "frac_reward_zero_std": 0.75, "grad_norm": 0.26826000213623047, "learning_rate": 1e-06, "loss": -0.0044, "num_tokens": 776435264.0, "reward": 0.6640625, "reward_std": 0.10771197080612183, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 1985, "tools/generated_tokens": 4633.0, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.6875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1177.06640625, "completions/mean_terminated_length": 1166.7391357421875, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 0.07640601322054863, "epoch": 0.3384241804588153, "frac_reward_zero_std": 0.4375, "grad_norm": 0.33291471004486084, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 776809569.0, "reward": 0.72265625, "reward_std": 0.22209003567695618, "rewards/simpleverify_reward/mean": 0.72265625, "rewards/simpleverify_reward/std": 0.4485645890235901, "step": 1986, "tools/generated_tokens": 3593.06640625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1796875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1207.71875, "completions/mean_terminated_length": 1201.1024169921875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.08422368997707963, "epoch": 0.3385945853835176, "frac_reward_zero_std": 0.375, "grad_norm": 0.43809786438941956, "learning_rate": 1e-06, "loss": 0.0106, "num_tokens": 777207881.0, "reward": 0.5234375, "reward_std": 0.2512268126010895, "rewards/simpleverify_reward/mean": 0.5234375, "rewards/simpleverify_reward/std": 0.5004287362098694, "step": 1987, "tools/generated_tokens": 4343.734375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.53125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2047.0, "completions/mean_length": 1184.2578125, "completions/mean_terminated_length": 1180.87060546875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.08154094265773892, "epoch": 0.3387649903082199, "frac_reward_zero_std": 0.4375, "grad_norm": 0.36457958817481995, "learning_rate": 1e-06, "loss": 0.0093, "num_tokens": 777587851.0, "reward": 0.4375, "reward_std": 0.18353557586669922, "rewards/simpleverify_reward/mean": 0.4375, "rewards/simpleverify_reward/std": 0.49705013632774353, "step": 1988, "tools/generated_tokens": 4440.265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.58984375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2029.0, "completions/max_terminated_length": 2029.0, "completions/mean_length": 1207.5546875, "completions/mean_terminated_length": 1207.5546875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.09112601075321436, "epoch": 0.3389353952329222, "frac_reward_zero_std": 0.5, "grad_norm": 0.4066215455532074, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 777959145.0, "reward": 0.515625, "reward_std": 0.18912501633167267, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1989, "tools/generated_tokens": 3463.5546875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.1015625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1233.43359375, "completions/mean_terminated_length": 1213.884033203125, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.07600541086867452, "epoch": 0.33910580015762454, "frac_reward_zero_std": 0.5, "grad_norm": 0.4053618609905243, "learning_rate": 1e-06, "loss": 0.0006, "num_tokens": 778343448.0, "reward": 0.5546875, "reward_std": 0.19588851928710938, "rewards/simpleverify_reward/mean": 0.5546875, "rewards/simpleverify_reward/std": 0.49797385931015015, "step": 1990, "tools/generated_tokens": 4329.44921875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.51171875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1155.375, "completions/mean_terminated_length": 1151.8746337890625, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.07821630546823144, "epoch": 0.33927620508232686, "frac_reward_zero_std": 0.625, "grad_norm": 0.3181440830230713, "learning_rate": 1e-06, "loss": -0.0032, "num_tokens": 778722344.0, "reward": 0.6640625, "reward_std": 0.13896197080612183, "rewards/simpleverify_reward/mean": 0.6640625, "rewards/simpleverify_reward/std": 0.4732423722743988, "step": 1991, "tools/generated_tokens": 4163.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.46875, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1213.95703125, "completions/mean_terminated_length": 1210.6864013671875, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.07982334261760116, "epoch": 0.3394466100070292, "frac_reward_zero_std": 0.4375, "grad_norm": 0.34379884600639343, "learning_rate": 1e-06, "loss": 0.0055, "num_tokens": 779096653.0, "reward": 0.61328125, "reward_std": 0.19545848667621613, "rewards/simpleverify_reward/mean": 0.61328125, "rewards/simpleverify_reward/std": 0.4879522919654846, "step": 1992, "tools/generated_tokens": 3061.96484375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 0.90234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2045.0, "completions/mean_length": 1219.40234375, "completions/mean_terminated_length": 1216.1529541015625, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.08265806455165148, "epoch": 0.3396170149317315, "frac_reward_zero_std": 0.5625, "grad_norm": 0.3236895799636841, "learning_rate": 1e-06, "loss": 0.0014, "num_tokens": 779486436.0, "reward": 0.48828125, "reward_std": 0.16471801698207855, "rewards/simpleverify_reward/mean": 0.48828125, "rewards/simpleverify_reward/std": 0.5008418560028076, "step": 1993, "tools/generated_tokens": 4067.39453125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.390625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2046.0, "completions/mean_length": 1260.80859375, "completions/mean_terminated_length": 1245.1275634765625, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.08232436468824744, "epoch": 0.33978741985643385, "frac_reward_zero_std": 0.5, "grad_norm": 0.3282334506511688, "learning_rate": 1e-06, "loss": -0.0116, "num_tokens": 779879139.0, "reward": 0.6171875, "reward_std": 0.1963387131690979, "rewards/simpleverify_reward/mean": 0.6171875, "rewards/simpleverify_reward/std": 0.48702529072761536, "step": 1994, "tools/generated_tokens": 3684.8203125, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.18359375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2030.0, "completions/mean_length": 1104.921875, "completions/mean_terminated_length": 1089.9564208984375, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.07938217651098967, "epoch": 0.3399578247811362, "frac_reward_zero_std": 0.4375, "grad_norm": 0.6312114596366882, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 780242703.0, "reward": 0.546875, "reward_std": 0.20750631392002106, "rewards/simpleverify_reward/mean": 0.546875, "rewards/simpleverify_reward/std": 0.4987730085849762, "step": 1995, "tools/generated_tokens": 4488.9296875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.65234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2048.0, "completions/max_terminated_length": 2048.0, "completions/mean_length": 1165.47265625, "completions/mean_terminated_length": 1165.47265625, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.09540707897394896, "epoch": 0.3401282297058385, "frac_reward_zero_std": 0.6875, "grad_norm": 0.27757757902145386, "learning_rate": 1e-06, "loss": 0.0108, "num_tokens": 780615128.0, "reward": 0.51953125, "reward_std": 0.0924195945262909, "rewards/simpleverify_reward/mean": 0.51953125, "rewards/simpleverify_reward/std": 0.5005971193313599, "step": 1996, "tools/generated_tokens": 3957.47265625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.36328125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2014.0, "completions/mean_length": 1177.4609375, "completions/mean_terminated_length": 1174.047119140625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.09043654752895236, "epoch": 0.34029863463054083, "frac_reward_zero_std": 0.5, "grad_norm": 0.5068737268447876, "learning_rate": 1e-06, "loss": -0.0109, "num_tokens": 780981022.0, "reward": 0.515625, "reward_std": 0.1857442855834961, "rewards/simpleverify_reward/mean": 0.515625, "rewards/simpleverify_reward/std": 0.5007347464561462, "step": 1997, "tools/generated_tokens": 3945.5390625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.3515625, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 2048.0, "completions/max_terminated_length": 2043.0, "completions/mean_length": 1141.578125, "completions/mean_terminated_length": 1138.0235595703125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.09700705157592893, "epoch": 0.34046903955524316, "frac_reward_zero_std": 0.5, "grad_norm": 0.41844499111175537, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 781363410.0, "reward": 0.390625, "reward_std": 0.23039758205413818, "rewards/simpleverify_reward/mean": 0.390625, "rewards/simpleverify_reward/std": 0.48884621262550354, "step": 1998, "tools/generated_tokens": 4229.59375, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.5078125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 2048.0, "completions/max_terminated_length": 2033.0, "completions/mean_length": 1154.625, "completions/mean_terminated_length": 1136.8287353515625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.10325901303440332, "epoch": 0.3406394444799455, "frac_reward_zero_std": 0.375, "grad_norm": 0.36369413137435913, "learning_rate": 1e-06, "loss": -0.0193, "num_tokens": 781740514.0, "reward": 0.54296875, "reward_std": 0.23490957915782928, "rewards/simpleverify_reward/mean": 0.54296875, "rewards/simpleverify_reward/std": 0.4991260766983032, "step": 1999, "tools/generated_tokens": 4130.625, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.453125, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 2048.0, "completions/max_terminated_length": 2039.0, "completions/mean_length": 1224.47265625, "completions/mean_terminated_length": 1214.70751953125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.09362190729007125, "epoch": 0.3408098494046478, "frac_reward_zero_std": 0.4375, "grad_norm": 0.4761008620262146, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 782128347.0, "reward": 0.59375, "reward_std": 0.21306806802749634, "rewards/simpleverify_reward/mean": 0.59375, "rewards/simpleverify_reward/std": 0.49209436774253845, "step": 2000, "tools/generated_tokens": 3584.48046875, "tools/num_python": 0.0, "tools/num_python_exec_error": 0.0, "tools/num_retrieval": 0.0, "tools/num_retriever_exec_error": 0.0, "tools/num_saving": 1.15234375, "tools/num_saving_exec_error": 0.0, "tools/num_saving_forced": 0.0, "tools/num_saving_invalid_use": 0.0, "tools/num_tool_detect_error": 0.0 }, { "epoch": 0.3408098494046478, "step": 2000, "total_flos": 0.0, "train_loss": 0.00040604471566621215, "train_runtime": 84291.6987, "train_samples_per_second": 6.074, "train_steps_per_second": 0.024 } ], "logging_steps": 1, "max_steps": 2000, "num_input_tokens_seen": 782128347, "num_train_epochs": 1, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }