| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 5.167958656330749, |
| "eval_steps": 500, |
| "global_step": 2000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 1497.2, |
| "completions/max_terminated_length": 1065.7, |
| "completions/mean_length": 135.1234375, |
| "completions/mean_terminated_length": 126.72116470336914, |
| "completions/min_length": 21.6, |
| "completions/min_terminated_length": 21.6, |
| "entropy": 0.18923678908031433, |
| "epoch": 0.025839793281653745, |
| "frac_reward_zero_std": 0.56875, |
| "grad_norm": 0.3984375, |
| "learning_rate": 9.9991e-06, |
| "loss": -0.0161, |
| "num_tokens": 660078.0, |
| "reward": 0.11468750350177288, |
| "reward_std": 0.1423981124535203, |
| "rewards/reward_accuracy/mean": 0.025, |
| "rewards/reward_accuracy/std": 0.13969720751047135, |
| "rewards/reward_format/mean": 0.08968750275671482, |
| "rewards/reward_format/std": 0.014365329034626484, |
| "sampling/importance_sampling_ratio/max": 2.344062077999115, |
| "sampling/importance_sampling_ratio/mean": 0.9256866633892059, |
| "sampling/importance_sampling_ratio/min": 0.08241417221724986, |
| "sampling/sampling_logp_difference/max": 0.5646095037460327, |
| "sampling/sampling_logp_difference/mean": 0.01094961455091834, |
| "step": 10, |
| "step_time": 28.97542249551043 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 1711.4, |
| "completions/max_terminated_length": 906.0, |
| "completions/mean_length": 86.70234375, |
| "completions/mean_terminated_length": 78.15720977783204, |
| "completions/min_length": 26.3, |
| "completions/min_terminated_length": 26.3, |
| "entropy": 0.12925479856785388, |
| "epoch": 0.05167958656330749, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.625, |
| "learning_rate": 9.9981e-06, |
| "loss": -0.0058, |
| "num_tokens": 1257745.0, |
| "reward": 0.1333203211426735, |
| "reward_std": 0.156949517223984, |
| "rewards/reward_accuracy/mean": 0.03359375, |
| "rewards/reward_accuracy/std": 0.1564072363078594, |
| "rewards/reward_format/mean": 0.09972656443715096, |
| "rewards/reward_format/std": 0.0021992066875100138, |
| "sampling/importance_sampling_ratio/max": 2.1037406802177427, |
| "sampling/importance_sampling_ratio/mean": 0.9739229261875153, |
| "sampling/importance_sampling_ratio/min": 0.09216372857335955, |
| "sampling/sampling_logp_difference/max": 0.5292048215866089, |
| "sampling/sampling_logp_difference/mean": 0.008396762236952782, |
| "step": 20, |
| "step_time": 32.92067137649283 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 502.4, |
| "completions/max_terminated_length": 502.4, |
| "completions/mean_length": 58.19609375, |
| "completions/mean_terminated_length": 58.19609375, |
| "completions/min_length": 30.7, |
| "completions/min_terminated_length": 30.7, |
| "entropy": 0.07240330665372312, |
| "epoch": 0.07751937984496124, |
| "frac_reward_zero_std": 0.83125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.997100000000001e-06, |
| "loss": -0.0076, |
| "num_tokens": 1813684.0, |
| "reward": 0.15058594346046447, |
| "reward_std": 0.19093237966299056, |
| "rewards/reward_accuracy/mean": 0.05078125, |
| "rewards/reward_accuracy/std": 0.19086164534091948, |
| "rewards/reward_format/mean": 0.09980468899011612, |
| "rewards/reward_format/std": 0.001948359701782465, |
| "sampling/importance_sampling_ratio/max": 1.7179017901420592, |
| "sampling/importance_sampling_ratio/mean": 0.9887637078762055, |
| "sampling/importance_sampling_ratio/min": 0.2173505738377571, |
| "sampling/sampling_logp_difference/max": 0.45053328275680543, |
| "sampling/sampling_logp_difference/mean": 0.0049190066754817964, |
| "step": 30, |
| "step_time": 11.014861900964751 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 286.6, |
| "completions/max_terminated_length": 286.6, |
| "completions/mean_length": 55.4109375, |
| "completions/mean_terminated_length": 55.4109375, |
| "completions/min_length": 28.1, |
| "completions/min_terminated_length": 28.1, |
| "entropy": 0.06418112071696669, |
| "epoch": 0.10335917312661498, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.16015625, |
| "learning_rate": 9.9961e-06, |
| "loss": -0.0019, |
| "num_tokens": 2371810.0, |
| "reward": 0.13515625894069672, |
| "reward_std": 0.17917687818408012, |
| "rewards/reward_accuracy/mean": 0.03515625, |
| "rewards/reward_accuracy/std": 0.17917687818408012, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6236321449279785, |
| "sampling/importance_sampling_ratio/mean": 0.9977541863918304, |
| "sampling/importance_sampling_ratio/min": 0.4504877131432295, |
| "sampling/sampling_logp_difference/max": 0.37731835842132566, |
| "sampling/sampling_logp_difference/mean": 0.003698127483949065, |
| "step": 40, |
| "step_time": 8.27236980102025 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 488.6, |
| "completions/max_terminated_length": 488.6, |
| "completions/mean_length": 54.08515625, |
| "completions/mean_terminated_length": 54.08515625, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.0624315386870876, |
| "epoch": 0.12919896640826872, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.494140625, |
| "learning_rate": 9.9951e-06, |
| "loss": -0.0012, |
| "num_tokens": 2920847.0, |
| "reward": 0.15617188513278962, |
| "reward_std": 0.21790467202663422, |
| "rewards/reward_accuracy/mean": 0.05625, |
| "rewards/reward_accuracy/std": 0.21787354350090027, |
| "rewards/reward_format/mean": 0.09992187619209289, |
| "rewards/reward_format/std": 0.00088388342410326, |
| "sampling/importance_sampling_ratio/max": 1.6224091172218322, |
| "sampling/importance_sampling_ratio/mean": 1.000568687915802, |
| "sampling/importance_sampling_ratio/min": 0.4336530029773712, |
| "sampling/sampling_logp_difference/max": 0.4461843013763428, |
| "sampling/sampling_logp_difference/mean": 0.004137374833226204, |
| "step": 50, |
| "step_time": 10.70177974156104 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 693.4, |
| "completions/max_terminated_length": 355.3, |
| "completions/mean_length": 55.959375, |
| "completions/mean_terminated_length": 53.123978805541995, |
| "completions/min_length": 25.8, |
| "completions/min_terminated_length": 25.8, |
| "entropy": 0.046299845934845506, |
| "epoch": 0.15503875968992248, |
| "frac_reward_zero_std": 0.86875, |
| "grad_norm": 0.373046875, |
| "learning_rate": 9.994100000000001e-06, |
| "loss": -0.0016, |
| "num_tokens": 3474891.0, |
| "reward": 0.2109375149011612, |
| "reward_std": 0.2950064606964588, |
| "rewards/reward_accuracy/mean": 0.1109375, |
| "rewards/reward_accuracy/std": 0.2950064606964588, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6936934351921082, |
| "sampling/importance_sampling_ratio/mean": 0.9957539916038514, |
| "sampling/importance_sampling_ratio/min": 0.3843412309885025, |
| "sampling/sampling_logp_difference/max": 0.4427079796791077, |
| "sampling/sampling_logp_difference/mean": 0.002955483319237828, |
| "step": 60, |
| "step_time": 15.37540449462831 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 188.6, |
| "completions/max_terminated_length": 188.6, |
| "completions/mean_length": 47.24375, |
| "completions/mean_terminated_length": 47.24375, |
| "completions/min_length": 26.3, |
| "completions/min_terminated_length": 26.3, |
| "entropy": 0.030355607363162562, |
| "epoch": 0.18087855297157623, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.68359375, |
| "learning_rate": 9.993100000000001e-06, |
| "loss": 0.0017, |
| "num_tokens": 4010627.0, |
| "reward": 0.25773439556360245, |
| "reward_std": 0.34589259549975393, |
| "rewards/reward_accuracy/mean": 0.1578125, |
| "rewards/reward_accuracy/std": 0.34583136066794395, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.3856260657310486, |
| "sampling/importance_sampling_ratio/mean": 0.9955059945583343, |
| "sampling/importance_sampling_ratio/min": 0.5913065433502197, |
| "sampling/sampling_logp_difference/max": 0.3812487363815308, |
| "sampling/sampling_logp_difference/mean": 0.002028863865416497, |
| "step": 70, |
| "step_time": 6.559066118043847 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.8, |
| "completions/max_terminated_length": 72.8, |
| "completions/mean_length": 47.3734375, |
| "completions/mean_terminated_length": 47.3734375, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.022748721425887197, |
| "epoch": 0.20671834625322996, |
| "frac_reward_zero_std": 0.86875, |
| "grad_norm": 0.82421875, |
| "learning_rate": 9.9921e-06, |
| "loss": -0.001, |
| "num_tokens": 4549993.0, |
| "reward": 0.3390625208616257, |
| "reward_std": 0.4028328523039818, |
| "rewards/reward_accuracy/mean": 0.2390625, |
| "rewards/reward_accuracy/std": 0.4028328493237495, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4024048686027526, |
| "sampling/importance_sampling_ratio/mean": 0.998824292421341, |
| "sampling/importance_sampling_ratio/min": 0.7334581792354584, |
| "sampling/sampling_logp_difference/max": 0.3834665149450302, |
| "sampling/sampling_logp_difference/mean": 0.0015821842709556222, |
| "step": 80, |
| "step_time": 4.983648956846446 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 111.7, |
| "completions/max_terminated_length": 111.7, |
| "completions/mean_length": 49.1859375, |
| "completions/mean_terminated_length": 49.1859375, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.02391942385584116, |
| "epoch": 0.23255813953488372, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.1328125, |
| "learning_rate": 9.991100000000002e-06, |
| "loss": 0.0017, |
| "num_tokens": 5095735.0, |
| "reward": 0.25695314407348635, |
| "reward_std": 0.3529519110918045, |
| "rewards/reward_accuracy/mean": 0.15703125, |
| "rewards/reward_accuracy/std": 0.3529082477092743, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.5546794891357423, |
| "sampling/importance_sampling_ratio/mean": 0.9947991371154785, |
| "sampling/importance_sampling_ratio/min": 0.6016295373439788, |
| "sampling/sampling_logp_difference/max": 0.4450272798538208, |
| "sampling/sampling_logp_difference/mean": 0.0018798474338836968, |
| "step": 90, |
| "step_time": 5.48268072286155 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 96.1, |
| "completions/max_terminated_length": 96.1, |
| "completions/mean_length": 50.0375, |
| "completions/mean_terminated_length": 50.0375, |
| "completions/min_length": 25.7, |
| "completions/min_terminated_length": 25.7, |
| "entropy": 0.026928382460027933, |
| "epoch": 0.25839793281653745, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.62890625, |
| "learning_rate": 9.990100000000001e-06, |
| "loss": -0.0014, |
| "num_tokens": 5643439.0, |
| "reward": 0.24062501788139343, |
| "reward_std": 0.33786573708057405, |
| "rewards/reward_accuracy/mean": 0.140625, |
| "rewards/reward_accuracy/std": 0.33786573708057405, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5223956823348999, |
| "sampling/importance_sampling_ratio/mean": 0.9991921365261078, |
| "sampling/importance_sampling_ratio/min": 0.6099381119012832, |
| "sampling/sampling_logp_difference/max": 0.459058940410614, |
| "sampling/sampling_logp_difference/mean": 0.0020969793782569467, |
| "step": 100, |
| "step_time": 5.324377598776482 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 82.8, |
| "completions/max_terminated_length": 82.8, |
| "completions/mean_length": 52.5265625, |
| "completions/mean_terminated_length": 52.5265625, |
| "completions/min_length": 28.6, |
| "completions/min_terminated_length": 28.6, |
| "entropy": 0.023515025299275294, |
| "epoch": 0.2842377260981912, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.7265625, |
| "learning_rate": 9.9891e-06, |
| "loss": 0.0081, |
| "num_tokens": 6199873.0, |
| "reward": 0.21640626788139344, |
| "reward_std": 0.2975174032151699, |
| "rewards/reward_accuracy/mean": 0.11640625, |
| "rewards/reward_accuracy/std": 0.2975174032151699, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6385587811470033, |
| "sampling/importance_sampling_ratio/mean": 1.0029362618923188, |
| "sampling/importance_sampling_ratio/min": 0.6264503121376037, |
| "sampling/sampling_logp_difference/max": 0.4787415385246277, |
| "sampling/sampling_logp_difference/mean": 0.0019316444988362492, |
| "step": 110, |
| "step_time": 5.198969352827407 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.0, |
| "completions/max_terminated_length": 74.0, |
| "completions/mean_length": 50.265625, |
| "completions/mean_terminated_length": 50.265625, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.01742738600296434, |
| "epoch": 0.31007751937984496, |
| "frac_reward_zero_std": 0.91875, |
| "grad_norm": 0.6953125, |
| "learning_rate": 9.9881e-06, |
| "loss": 0.0006, |
| "num_tokens": 6750293.0, |
| "reward": 0.22265626713633538, |
| "reward_std": 0.2943296328186989, |
| "rewards/reward_accuracy/mean": 0.12265625, |
| "rewards/reward_accuracy/std": 0.29432962983846667, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6133869290351868, |
| "sampling/importance_sampling_ratio/mean": 1.0030302047729491, |
| "sampling/importance_sampling_ratio/min": 0.5885300070047379, |
| "sampling/sampling_logp_difference/max": 0.48409855365753174, |
| "sampling/sampling_logp_difference/mean": 0.0015391250140964984, |
| "step": 120, |
| "step_time": 4.97823460586369 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 94.4, |
| "completions/max_terminated_length": 94.4, |
| "completions/mean_length": 49.34296875, |
| "completions/mean_terminated_length": 49.34296875, |
| "completions/min_length": 25.3, |
| "completions/min_terminated_length": 25.3, |
| "entropy": 0.017413872395991348, |
| "epoch": 0.3359173126614987, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.412109375, |
| "learning_rate": 9.987100000000001e-06, |
| "loss": -0.0018, |
| "num_tokens": 7297820.0, |
| "reward": 0.31875002235174177, |
| "reward_std": 0.40157595574855803, |
| "rewards/reward_accuracy/mean": 0.21875, |
| "rewards/reward_accuracy/std": 0.4015759527683258, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3913645267486572, |
| "sampling/importance_sampling_ratio/mean": 1.0026760876178742, |
| "sampling/importance_sampling_ratio/min": 0.6181297466158867, |
| "sampling/sampling_logp_difference/max": 0.5119511663913727, |
| "sampling/sampling_logp_difference/mean": 0.0015221343259327114, |
| "step": 130, |
| "step_time": 5.30160520542413 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 79.3, |
| "completions/max_terminated_length": 79.3, |
| "completions/mean_length": 50.76796875, |
| "completions/mean_terminated_length": 50.76796875, |
| "completions/min_length": 27.6, |
| "completions/min_terminated_length": 27.6, |
| "entropy": 0.01753365322656464, |
| "epoch": 0.36175710594315247, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 1.7890625, |
| "learning_rate": 9.9861e-06, |
| "loss": -0.0005, |
| "num_tokens": 7848947.0, |
| "reward": 0.2937500163912773, |
| "reward_std": 0.3613394796848297, |
| "rewards/reward_accuracy/mean": 0.19375, |
| "rewards/reward_accuracy/std": 0.3613394796848297, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.622401511669159, |
| "sampling/importance_sampling_ratio/mean": 1.0011253476142883, |
| "sampling/importance_sampling_ratio/min": 0.621192216873169, |
| "sampling/sampling_logp_difference/max": 0.48284724950790403, |
| "sampling/sampling_logp_difference/mean": 0.0017386986641213299, |
| "step": 140, |
| "step_time": 5.085251715569757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.0, |
| "completions/max_terminated_length": 74.0, |
| "completions/mean_length": 50.25390625, |
| "completions/mean_terminated_length": 50.25390625, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.014563843313953839, |
| "epoch": 0.3875968992248062, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 1.2421875, |
| "learning_rate": 9.9851e-06, |
| "loss": 0.0006, |
| "num_tokens": 8399264.0, |
| "reward": 0.31406252086162567, |
| "reward_std": 0.39328832775354383, |
| "rewards/reward_accuracy/mean": 0.2140625, |
| "rewards/reward_accuracy/std": 0.3932883247733116, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4871245980262757, |
| "sampling/importance_sampling_ratio/mean": 1.0000156581401825, |
| "sampling/importance_sampling_ratio/min": 0.648833030462265, |
| "sampling/sampling_logp_difference/max": 0.48496219515800476, |
| "sampling/sampling_logp_difference/mean": 0.001299281616229564, |
| "step": 150, |
| "step_time": 4.996830363851041 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 433.8, |
| "completions/max_terminated_length": 76.6, |
| "completions/mean_length": 54.23828125, |
| "completions/mean_terminated_length": 51.41728591918945, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.01427922679868061, |
| "epoch": 0.4134366925064599, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.6328125, |
| "learning_rate": 9.984100000000002e-06, |
| "loss": 0.0022, |
| "num_tokens": 8957585.0, |
| "reward": 0.29449220597743986, |
| "reward_std": 0.37526765614748003, |
| "rewards/reward_accuracy/mean": 0.19453125, |
| "rewards/reward_accuracy/std": 0.3752377137541771, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.4751390099525452, |
| "sampling/importance_sampling_ratio/mean": 0.9972580552101136, |
| "sampling/importance_sampling_ratio/min": 0.5495553106069565, |
| "sampling/sampling_logp_difference/max": 0.6277500987052917, |
| "sampling/sampling_logp_difference/mean": 0.001398616493679583, |
| "step": 160, |
| "step_time": 11.88281731097959 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 80.6, |
| "completions/max_terminated_length": 80.6, |
| "completions/mean_length": 51.99140625, |
| "completions/mean_terminated_length": 51.99140625, |
| "completions/min_length": 25.7, |
| "completions/min_terminated_length": 25.7, |
| "entropy": 0.012813940588966944, |
| "epoch": 0.4392764857881137, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.625, |
| "learning_rate": 9.983100000000001e-06, |
| "loss": 0.0031, |
| "num_tokens": 9514750.0, |
| "reward": 0.29140626788139345, |
| "reward_std": 0.37911527752876284, |
| "rewards/reward_accuracy/mean": 0.19140625, |
| "rewards/reward_accuracy/std": 0.37911527752876284, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4252223730087281, |
| "sampling/importance_sampling_ratio/mean": 0.9976849794387818, |
| "sampling/importance_sampling_ratio/min": 0.6736114114522934, |
| "sampling/sampling_logp_difference/max": 0.43279322385787966, |
| "sampling/sampling_logp_difference/mean": 0.0011501138738822191, |
| "step": 170, |
| "step_time": 5.144311515870504 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.0813149128807708e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.0813149128807708e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 88.2, |
| "completions/max_terminated_length": 88.2, |
| "completions/mean_length": 50.571875, |
| "completions/mean_terminated_length": 50.571875, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.01384837388759479, |
| "epoch": 0.46511627906976744, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9821e-06, |
| "loss": 0.0015, |
| "num_tokens": 10066578.0, |
| "reward": 0.28359376788139345, |
| "reward_std": 0.37317481338977815, |
| "rewards/reward_accuracy/mean": 0.18359375, |
| "rewards/reward_accuracy/std": 0.37317481338977815, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.528812623023987, |
| "sampling/importance_sampling_ratio/mean": 1.0018202900886535, |
| "sampling/importance_sampling_ratio/min": 0.6639125227928162, |
| "sampling/sampling_logp_difference/max": 0.3892782121896744, |
| "sampling/sampling_logp_difference/mean": 0.0011624884384218604, |
| "step": 180, |
| "step_time": 5.280760875297711 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 82.8, |
| "completions/max_terminated_length": 82.8, |
| "completions/mean_length": 51.328125, |
| "completions/mean_terminated_length": 51.328125, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.016378523065941408, |
| "epoch": 0.4909560723514212, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.981100000000002e-06, |
| "loss": -0.0006, |
| "num_tokens": 10620150.0, |
| "reward": 0.27109376788139344, |
| "reward_std": 0.3516123160719872, |
| "rewards/reward_accuracy/mean": 0.17109375, |
| "rewards/reward_accuracy/std": 0.3516123160719872, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4416401028633117, |
| "sampling/importance_sampling_ratio/mean": 0.995251727104187, |
| "sampling/importance_sampling_ratio/min": 0.6335062682628632, |
| "sampling/sampling_logp_difference/max": 0.4753629148006439, |
| "sampling/sampling_logp_difference/mean": 0.0016480274382047356, |
| "step": 190, |
| "step_time": 5.101334654120729 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 108.5, |
| "completions/max_terminated_length": 108.5, |
| "completions/mean_length": 48.29609375, |
| "completions/mean_terminated_length": 48.29609375, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.013529703120002522, |
| "epoch": 0.5167958656330749, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.98828125, |
| "learning_rate": 9.9801e-06, |
| "loss": -0.0008, |
| "num_tokens": 11161137.0, |
| "reward": 0.303085957467556, |
| "reward_std": 0.3985190987586975, |
| "rewards/reward_accuracy/mean": 0.203125, |
| "rewards/reward_accuracy/std": 0.39850133955478667, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.6519379138946533, |
| "sampling/importance_sampling_ratio/mean": 0.9988101899623871, |
| "sampling/importance_sampling_ratio/min": 0.6785599738359451, |
| "sampling/sampling_logp_difference/max": 0.5063596189022064, |
| "sampling/sampling_logp_difference/mean": 0.0014554417924955488, |
| "step": 200, |
| "step_time": 5.448744368040934 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 93.7, |
| "completions/max_terminated_length": 93.7, |
| "completions/mean_length": 50.41640625, |
| "completions/mean_terminated_length": 50.41640625, |
| "completions/min_length": 26.3, |
| "completions/min_terminated_length": 26.3, |
| "entropy": 0.01635951530188322, |
| "epoch": 0.5426356589147286, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.33203125, |
| "learning_rate": 9.9791e-06, |
| "loss": -0.0043, |
| "num_tokens": 11711462.0, |
| "reward": 0.3148437738418579, |
| "reward_std": 0.4010762542486191, |
| "rewards/reward_accuracy/mean": 0.21484375, |
| "rewards/reward_accuracy/std": 0.4010762542486191, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8109891533851623, |
| "sampling/importance_sampling_ratio/mean": 1.0060723543167114, |
| "sampling/importance_sampling_ratio/min": 0.6853438675403595, |
| "sampling/sampling_logp_difference/max": 0.4793629288673401, |
| "sampling/sampling_logp_difference/mean": 0.0014200011151842772, |
| "step": 210, |
| "step_time": 5.245541226840578 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 81.0, |
| "completions/max_terminated_length": 81.0, |
| "completions/mean_length": 49.71796875, |
| "completions/mean_terminated_length": 49.71796875, |
| "completions/min_length": 27.6, |
| "completions/min_terminated_length": 27.6, |
| "entropy": 0.011892485243151896, |
| "epoch": 0.5684754521963824, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 1.1171875, |
| "learning_rate": 9.9781e-06, |
| "loss": 0.0022, |
| "num_tokens": 12259589.0, |
| "reward": 0.2882812708616257, |
| "reward_std": 0.3879868805408478, |
| "rewards/reward_accuracy/mean": 0.18828125, |
| "rewards/reward_accuracy/std": 0.3879868805408478, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4753874659538269, |
| "sampling/importance_sampling_ratio/mean": 0.9999773561954498, |
| "sampling/importance_sampling_ratio/min": 0.6720856726169586, |
| "sampling/sampling_logp_difference/max": 0.449326890707016, |
| "sampling/sampling_logp_difference/mean": 0.001138064614497125, |
| "step": 220, |
| "step_time": 5.138022253941744 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 78.8, |
| "completions/max_terminated_length": 78.8, |
| "completions/mean_length": 52.6890625, |
| "completions/mean_terminated_length": 52.6890625, |
| "completions/min_length": 28.3, |
| "completions/min_terminated_length": 28.3, |
| "entropy": 0.010768312203435926, |
| "epoch": 0.5943152454780362, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.977100000000001e-06, |
| "loss": 0.0003, |
| "num_tokens": 12818175.0, |
| "reward": 0.25234376788139345, |
| "reward_std": 0.34311832785606383, |
| "rewards/reward_accuracy/mean": 0.15234375, |
| "rewards/reward_accuracy/std": 0.3431183248758316, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4662445902824401, |
| "sampling/importance_sampling_ratio/mean": 1.002385139465332, |
| "sampling/importance_sampling_ratio/min": 0.6570447474718094, |
| "sampling/sampling_logp_difference/max": 0.39059698581695557, |
| "sampling/sampling_logp_difference/mean": 0.0010319790337234736, |
| "step": 230, |
| "step_time": 5.074341614730656 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 83.8, |
| "completions/max_terminated_length": 83.8, |
| "completions/mean_length": 49.30703125, |
| "completions/mean_terminated_length": 49.30703125, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.010051963504520246, |
| "epoch": 0.6201550387596899, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.9921875, |
| "learning_rate": 9.976100000000001e-06, |
| "loss": -0.0007, |
| "num_tokens": 13364552.0, |
| "reward": 0.2835937663912773, |
| "reward_std": 0.363788990676403, |
| "rewards/reward_accuracy/mean": 0.18359375, |
| "rewards/reward_accuracy/std": 0.363788990676403, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.473850131034851, |
| "sampling/importance_sampling_ratio/mean": 1.0024731397628783, |
| "sampling/importance_sampling_ratio/min": 0.6528496980667114, |
| "sampling/sampling_logp_difference/max": 0.4450443685054779, |
| "sampling/sampling_logp_difference/mean": 0.0009845186024904252, |
| "step": 240, |
| "step_time": 5.23160491886083 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.9, |
| "completions/max_terminated_length": 72.9, |
| "completions/mean_length": 48.58046875, |
| "completions/mean_terminated_length": 48.58046875, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.010150269603764172, |
| "epoch": 0.6459948320413437, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9751e-06, |
| "loss": -0.0006, |
| "num_tokens": 13906655.0, |
| "reward": 0.3226562708616257, |
| "reward_std": 0.39717224091291425, |
| "rewards/reward_accuracy/mean": 0.22265625, |
| "rewards/reward_accuracy/std": 0.397172237932682, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6466259360313416, |
| "sampling/importance_sampling_ratio/mean": 1.001248264312744, |
| "sampling/importance_sampling_ratio/min": 0.6828350961208344, |
| "sampling/sampling_logp_difference/max": 0.48007344007492064, |
| "sampling/sampling_logp_difference/mean": 0.0010736658063251526, |
| "step": 250, |
| "step_time": 4.926319691142998 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.8, |
| "completions/max_terminated_length": 72.8, |
| "completions/mean_length": 50.49140625, |
| "completions/mean_terminated_length": 50.49140625, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.013893579969590064, |
| "epoch": 0.6718346253229974, |
| "frac_reward_zero_std": 0.90625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.974100000000002e-06, |
| "loss": 0.0028, |
| "num_tokens": 14455812.0, |
| "reward": 0.2958984553813934, |
| "reward_std": 0.3734433650970459, |
| "rewards/reward_accuracy/mean": 0.19609375, |
| "rewards/reward_accuracy/std": 0.37331042587757113, |
| "rewards/reward_format/mean": 0.09980468899011612, |
| "rewards/reward_format/std": 0.0009725249372422695, |
| "sampling/importance_sampling_ratio/max": 1.7765609860420226, |
| "sampling/importance_sampling_ratio/mean": 1.001291173696518, |
| "sampling/importance_sampling_ratio/min": 0.5914752334356308, |
| "sampling/sampling_logp_difference/max": 0.6264029204845428, |
| "sampling/sampling_logp_difference/mean": 0.0014127150527201593, |
| "step": 260, |
| "step_time": 4.918769885203801 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 77.9, |
| "completions/max_terminated_length": 77.9, |
| "completions/mean_length": 51.31328125, |
| "completions/mean_terminated_length": 51.31328125, |
| "completions/min_length": 27.7, |
| "completions/min_terminated_length": 27.7, |
| "entropy": 0.013378887067665346, |
| "epoch": 0.6976744186046512, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 1.4765625, |
| "learning_rate": 9.973100000000001e-06, |
| "loss": -0.0022, |
| "num_tokens": 15008037.0, |
| "reward": 0.2826172083616257, |
| "reward_std": 0.3688479453325272, |
| "rewards/reward_accuracy/mean": 0.1828125, |
| "rewards/reward_accuracy/std": 0.3686841011047363, |
| "rewards/reward_format/mean": 0.09980468899011612, |
| "rewards/reward_format/std": 0.0009725249372422695, |
| "sampling/importance_sampling_ratio/max": 1.6567806482315064, |
| "sampling/importance_sampling_ratio/mean": 1.0012046992778778, |
| "sampling/importance_sampling_ratio/min": 0.5288221716880799, |
| "sampling/sampling_logp_difference/max": 0.6373290985822677, |
| "sampling/sampling_logp_difference/mean": 0.0014519813761580736, |
| "step": 270, |
| "step_time": 5.122667407104745 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 115.6, |
| "completions/max_terminated_length": 115.6, |
| "completions/mean_length": 52.3734375, |
| "completions/mean_terminated_length": 52.3734375, |
| "completions/min_length": 29.6, |
| "completions/min_terminated_length": 29.6, |
| "entropy": 0.016053864013520068, |
| "epoch": 0.7235142118863049, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.68359375, |
| "learning_rate": 9.9721e-06, |
| "loss": -0.0028, |
| "num_tokens": 15564491.0, |
| "reward": 0.3210937723517418, |
| "reward_std": 0.4064185917377472, |
| "rewards/reward_accuracy/mean": 0.22109375, |
| "rewards/reward_accuracy/std": 0.4064185917377472, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.619066023826599, |
| "sampling/importance_sampling_ratio/mean": 0.9984265387058258, |
| "sampling/importance_sampling_ratio/min": 0.5296623766422272, |
| "sampling/sampling_logp_difference/max": 0.639452052116394, |
| "sampling/sampling_logp_difference/mean": 0.0018036386871244758, |
| "step": 280, |
| "step_time": 5.587187921348959 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 137.9, |
| "completions/max_terminated_length": 137.9, |
| "completions/mean_length": 48.95625, |
| "completions/mean_terminated_length": 48.95625, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.014330488932318985, |
| "epoch": 0.7493540051679587, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 1.09375, |
| "learning_rate": 9.971100000000002e-06, |
| "loss": -0.0003, |
| "num_tokens": 16106131.0, |
| "reward": 0.331210957467556, |
| "reward_std": 0.396498441696167, |
| "rewards/reward_accuracy/mean": 0.23125, |
| "rewards/reward_accuracy/std": 0.39647083878517153, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.5090100765228271, |
| "sampling/importance_sampling_ratio/mean": 0.9996052205562591, |
| "sampling/importance_sampling_ratio/min": 0.536342154443264, |
| "sampling/sampling_logp_difference/max": 0.6359906375408173, |
| "sampling/sampling_logp_difference/mean": 0.0015164271870162338, |
| "step": 290, |
| "step_time": 5.954436889383942 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 207.2, |
| "completions/max_terminated_length": 207.2, |
| "completions/mean_length": 51.05546875, |
| "completions/mean_terminated_length": 51.05546875, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.011957520361465867, |
| "epoch": 0.7751937984496124, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 1.203125, |
| "learning_rate": 9.9701e-06, |
| "loss": 0.001, |
| "num_tokens": 16654746.0, |
| "reward": 0.3742187723517418, |
| "reward_std": 0.4325857847929001, |
| "rewards/reward_accuracy/mean": 0.27421875, |
| "rewards/reward_accuracy/std": 0.4325857847929001, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4730702757835388, |
| "sampling/importance_sampling_ratio/mean": 0.9996255934238434, |
| "sampling/importance_sampling_ratio/min": 0.711204880475998, |
| "sampling/sampling_logp_difference/max": 0.4132277011871338, |
| "sampling/sampling_logp_difference/mean": 0.0012113257267628796, |
| "step": 300, |
| "step_time": 6.919236557278782 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.6, |
| "completions/max_terminated_length": 70.6, |
| "completions/mean_length": 50.36796875, |
| "completions/mean_terminated_length": 50.36796875, |
| "completions/min_length": 28.6, |
| "completions/min_terminated_length": 28.6, |
| "entropy": 0.007573664350638864, |
| "epoch": 0.8010335917312662, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.6796875, |
| "learning_rate": 9.9691e-06, |
| "loss": -0.0046, |
| "num_tokens": 17205105.0, |
| "reward": 0.31953126937150955, |
| "reward_std": 0.3927790954709053, |
| "rewards/reward_accuracy/mean": 0.21953125, |
| "rewards/reward_accuracy/std": 0.39277909249067305, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7508309364318848, |
| "sampling/importance_sampling_ratio/mean": 1.0001774668693542, |
| "sampling/importance_sampling_ratio/min": 0.7208467185497284, |
| "sampling/sampling_logp_difference/max": 0.5344860315322876, |
| "sampling/sampling_logp_difference/mean": 0.0010599425324471668, |
| "step": 310, |
| "step_time": 4.924165843613446 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 98.7, |
| "completions/max_terminated_length": 98.7, |
| "completions/mean_length": 51.25078125, |
| "completions/mean_terminated_length": 51.25078125, |
| "completions/min_length": 26.7, |
| "completions/min_terminated_length": 26.7, |
| "entropy": 0.008398198919167044, |
| "epoch": 0.8268733850129198, |
| "frac_reward_zero_std": 0.91875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9681e-06, |
| "loss": 0.004, |
| "num_tokens": 17757842.0, |
| "reward": 0.3492187723517418, |
| "reward_std": 0.4237205654382706, |
| "rewards/reward_accuracy/mean": 0.24921875, |
| "rewards/reward_accuracy/std": 0.4237205654382706, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5355341553688049, |
| "sampling/importance_sampling_ratio/mean": 0.9954140424728394, |
| "sampling/importance_sampling_ratio/min": 0.6350526839494706, |
| "sampling/sampling_logp_difference/max": 0.5927676200866699, |
| "sampling/sampling_logp_difference/mean": 0.0012294158164877444, |
| "step": 320, |
| "step_time": 5.4075505347456785 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 79.2, |
| "completions/max_terminated_length": 79.2, |
| "completions/mean_length": 49.60625, |
| "completions/mean_terminated_length": 49.60625, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.007021298252220731, |
| "epoch": 0.8527131782945736, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 1.6015625, |
| "learning_rate": 9.9671e-06, |
| "loss": -0.0012, |
| "num_tokens": 18303298.0, |
| "reward": 0.29140626937150954, |
| "reward_std": 0.3775685355067253, |
| "rewards/reward_accuracy/mean": 0.19140625, |
| "rewards/reward_accuracy/std": 0.3775685355067253, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3837327361106873, |
| "sampling/importance_sampling_ratio/mean": 0.9984965920448303, |
| "sampling/importance_sampling_ratio/min": 0.6239041864871979, |
| "sampling/sampling_logp_difference/max": 0.5378094911575317, |
| "sampling/sampling_logp_difference/mean": 0.0010398205340607092, |
| "step": 330, |
| "step_time": 5.144762963009998 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.7, |
| "completions/max_terminated_length": 73.7, |
| "completions/mean_length": 51.57265625, |
| "completions/mean_terminated_length": 51.57265625, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.007028296844509896, |
| "epoch": 0.8785529715762274, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.384765625, |
| "learning_rate": 9.966100000000001e-06, |
| "loss": -0.001, |
| "num_tokens": 18857415.0, |
| "reward": 0.27105470448732377, |
| "reward_std": 0.3578765757381916, |
| "rewards/reward_accuracy/mean": 0.17109375, |
| "rewards/reward_accuracy/std": 0.35785746946930885, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.4823655247688294, |
| "sampling/importance_sampling_ratio/mean": 1.0007174551486968, |
| "sampling/importance_sampling_ratio/min": 0.7223353683948517, |
| "sampling/sampling_logp_difference/max": 0.4307470440864563, |
| "sampling/sampling_logp_difference/mean": 0.0008048571704421193, |
| "step": 340, |
| "step_time": 4.963697974942624 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.8, |
| "completions/max_terminated_length": 72.8, |
| "completions/mean_length": 51.58203125, |
| "completions/mean_terminated_length": 51.58203125, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.007315734790972783, |
| "epoch": 0.9043927648578811, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.2109375, |
| "learning_rate": 9.9651e-06, |
| "loss": -0.0005, |
| "num_tokens": 19412384.0, |
| "reward": 0.3234375223517418, |
| "reward_std": 0.3933984525501728, |
| "rewards/reward_accuracy/mean": 0.2234375, |
| "rewards/reward_accuracy/std": 0.3933984495699406, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5271214723587037, |
| "sampling/importance_sampling_ratio/mean": 0.9994868218898774, |
| "sampling/importance_sampling_ratio/min": 0.6041404277086257, |
| "sampling/sampling_logp_difference/max": 0.600176477432251, |
| "sampling/sampling_logp_difference/mean": 0.0010149589885259046, |
| "step": 350, |
| "step_time": 5.0378726598108186 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.8, |
| "completions/max_terminated_length": 72.8, |
| "completions/mean_length": 52.009375, |
| "completions/mean_terminated_length": 52.009375, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.00632761939559714, |
| "epoch": 0.9302325581395349, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.58203125, |
| "learning_rate": 9.964100000000002e-06, |
| "loss": 0.0009, |
| "num_tokens": 19969324.0, |
| "reward": 0.2843750178813934, |
| "reward_std": 0.3586010843515396, |
| "rewards/reward_accuracy/mean": 0.184375, |
| "rewards/reward_accuracy/std": 0.3586010843515396, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4012197256088257, |
| "sampling/importance_sampling_ratio/mean": 0.992496645450592, |
| "sampling/importance_sampling_ratio/min": 0.632025945186615, |
| "sampling/sampling_logp_difference/max": 0.506392240524292, |
| "sampling/sampling_logp_difference/mean": 0.0008945498440880329, |
| "step": 360, |
| "step_time": 5.0156942392000925 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 80.5, |
| "completions/max_terminated_length": 80.5, |
| "completions/mean_length": 51.30703125, |
| "completions/mean_terminated_length": 51.30703125, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.008037836640141904, |
| "epoch": 0.9560723514211886, |
| "frac_reward_zero_std": 0.91875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.963100000000001e-06, |
| "loss": 0.0014, |
| "num_tokens": 20522805.0, |
| "reward": 0.2937500193715096, |
| "reward_std": 0.36316741183400153, |
| "rewards/reward_accuracy/mean": 0.19375, |
| "rewards/reward_accuracy/std": 0.36316741183400153, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.584621798992157, |
| "sampling/importance_sampling_ratio/mean": 1.0039340436458588, |
| "sampling/importance_sampling_ratio/min": 0.6855102017521858, |
| "sampling/sampling_logp_difference/max": 0.45594470500946044, |
| "sampling/sampling_logp_difference/mean": 0.0009574506169883534, |
| "step": 370, |
| "step_time": 5.148022883501835 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 82.1, |
| "completions/max_terminated_length": 82.1, |
| "completions/mean_length": 50.69453125, |
| "completions/mean_terminated_length": 50.69453125, |
| "completions/min_length": 25.7, |
| "completions/min_terminated_length": 25.7, |
| "entropy": 0.007932509649253915, |
| "epoch": 0.9819121447028424, |
| "frac_reward_zero_std": 0.90625, |
| "grad_norm": 0.83203125, |
| "learning_rate": 9.9621e-06, |
| "loss": -0.002, |
| "num_tokens": 21072654.0, |
| "reward": 0.33750002086162567, |
| "reward_std": 0.4079906210303307, |
| "rewards/reward_accuracy/mean": 0.2375, |
| "rewards/reward_accuracy/std": 0.4079906210303307, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3957651257514954, |
| "sampling/importance_sampling_ratio/mean": 0.9992758750915527, |
| "sampling/importance_sampling_ratio/min": 0.6447268337011337, |
| "sampling/sampling_logp_difference/max": 0.5248695135116577, |
| "sampling/sampling_logp_difference/mean": 0.001043213886441663, |
| "step": 380, |
| "step_time": 5.10830789625179 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 76.0, |
| "completions/max_terminated_length": 76.0, |
| "completions/mean_length": 48.3, |
| "completions/mean_terminated_length": 48.3, |
| "completions/min_length": 27.9, |
| "completions/min_terminated_length": 27.9, |
| "entropy": 0.007299173749197507, |
| "epoch": 1.0077519379844961, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.961100000000002e-06, |
| "loss": 0.0022, |
| "num_tokens": 21613438.0, |
| "reward": 0.3226562708616257, |
| "reward_std": 0.3983902186155319, |
| "rewards/reward_accuracy/mean": 0.22265625, |
| "rewards/reward_accuracy/std": 0.3983902186155319, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4169399976730346, |
| "sampling/importance_sampling_ratio/mean": 0.9983580708503723, |
| "sampling/importance_sampling_ratio/min": 0.6847252905368805, |
| "sampling/sampling_logp_difference/max": 0.4400125086307526, |
| "sampling/sampling_logp_difference/mean": 0.000850645184982568, |
| "step": 390, |
| "step_time": 4.948761158855632 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.3, |
| "completions/max_terminated_length": 71.3, |
| "completions/mean_length": 49.903125, |
| "completions/mean_terminated_length": 49.903125, |
| "completions/min_length": 29.7, |
| "completions/min_terminated_length": 29.7, |
| "entropy": 0.00718860717388452, |
| "epoch": 1.0335917312661498, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9601e-06, |
| "loss": -0.003, |
| "num_tokens": 22163034.0, |
| "reward": 0.30781251937150955, |
| "reward_std": 0.36807855889201163, |
| "rewards/reward_accuracy/mean": 0.2078125, |
| "rewards/reward_accuracy/std": 0.3680785559117794, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5243940949440002, |
| "sampling/importance_sampling_ratio/mean": 0.9929626286029816, |
| "sampling/importance_sampling_ratio/min": 0.6144364804029465, |
| "sampling/sampling_logp_difference/max": 0.5474963486194611, |
| "sampling/sampling_logp_difference/mean": 0.0009605619037756696, |
| "step": 400, |
| "step_time": 5.020826336718164 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 116.3, |
| "completions/max_terminated_length": 116.3, |
| "completions/mean_length": 53.65859375, |
| "completions/mean_terminated_length": 53.65859375, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.008256974280084251, |
| "epoch": 1.0594315245478036, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.5234375, |
| "learning_rate": 9.959100000000001e-06, |
| "loss": 0.0028, |
| "num_tokens": 22723557.0, |
| "reward": 0.3507812708616257, |
| "reward_std": 0.4172993332147598, |
| "rewards/reward_accuracy/mean": 0.25078125, |
| "rewards/reward_accuracy/std": 0.4172993332147598, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3819338440895081, |
| "sampling/importance_sampling_ratio/mean": 0.9964124619960785, |
| "sampling/importance_sampling_ratio/min": 0.5652711063623428, |
| "sampling/sampling_logp_difference/max": 0.49788941740989684, |
| "sampling/sampling_logp_difference/mean": 0.0009068347775610164, |
| "step": 410, |
| "step_time": 5.70360893602483 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.4, |
| "completions/max_terminated_length": 73.4, |
| "completions/mean_length": 49.4921875, |
| "completions/mean_terminated_length": 49.4921875, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.007465429428702919, |
| "epoch": 1.0852713178294573, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.6953125, |
| "learning_rate": 9.9581e-06, |
| "loss": -0.0008, |
| "num_tokens": 23270219.0, |
| "reward": 0.32343751937150955, |
| "reward_std": 0.39878137707710265, |
| "rewards/reward_accuracy/mean": 0.2234375, |
| "rewards/reward_accuracy/std": 0.39878137707710265, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5955763697624206, |
| "sampling/importance_sampling_ratio/mean": 0.9970046758651734, |
| "sampling/importance_sampling_ratio/min": 0.6086857497692109, |
| "sampling/sampling_logp_difference/max": 0.7256279349327087, |
| "sampling/sampling_logp_difference/mean": 0.001200965212774463, |
| "step": 420, |
| "step_time": 5.01378038583789 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 94.8, |
| "completions/max_terminated_length": 94.8, |
| "completions/mean_length": 50.58203125, |
| "completions/mean_terminated_length": 50.58203125, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.007754007725452539, |
| "epoch": 1.1111111111111112, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.58984375, |
| "learning_rate": 9.9571e-06, |
| "loss": -0.0024, |
| "num_tokens": 23820036.0, |
| "reward": 0.34218752235174177, |
| "reward_std": 0.42006864547729494, |
| "rewards/reward_accuracy/mean": 0.2421875, |
| "rewards/reward_accuracy/std": 0.42006864249706266, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5857088565826416, |
| "sampling/importance_sampling_ratio/mean": 1.0015143930912018, |
| "sampling/importance_sampling_ratio/min": 0.709397941827774, |
| "sampling/sampling_logp_difference/max": 0.4979455888271332, |
| "sampling/sampling_logp_difference/mean": 0.0008148724708007648, |
| "step": 430, |
| "step_time": 5.328577196155675 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 77.7, |
| "completions/max_terminated_length": 77.7, |
| "completions/mean_length": 52.2265625, |
| "completions/mean_terminated_length": 52.2265625, |
| "completions/min_length": 24.7, |
| "completions/min_terminated_length": 24.7, |
| "entropy": 0.007619669740233803, |
| "epoch": 1.1369509043927648, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.453125, |
| "learning_rate": 9.956100000000001e-06, |
| "loss": -0.0015, |
| "num_tokens": 24378326.0, |
| "reward": 0.32031252086162565, |
| "reward_std": 0.3938863143324852, |
| "rewards/reward_accuracy/mean": 0.2203125, |
| "rewards/reward_accuracy/std": 0.3938863143324852, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.349657368659973, |
| "sampling/importance_sampling_ratio/mean": 0.9956431686878204, |
| "sampling/importance_sampling_ratio/min": 0.6350584745407104, |
| "sampling/sampling_logp_difference/max": 0.5541166543960572, |
| "sampling/sampling_logp_difference/mean": 0.0008658544247737154, |
| "step": 440, |
| "step_time": 5.133205491583794 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.1, |
| "completions/max_terminated_length": 71.1, |
| "completions/mean_length": 48.534375, |
| "completions/mean_terminated_length": 48.534375, |
| "completions/min_length": 26.3, |
| "completions/min_terminated_length": 26.3, |
| "entropy": 0.0064260311759426255, |
| "epoch": 1.1627906976744187, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.349609375, |
| "learning_rate": 9.9551e-06, |
| "loss": -0.0025, |
| "num_tokens": 24922018.0, |
| "reward": 0.3406250238418579, |
| "reward_std": 0.41922587156295776, |
| "rewards/reward_accuracy/mean": 0.240625, |
| "rewards/reward_accuracy/std": 0.41922587156295776, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.433572566509247, |
| "sampling/importance_sampling_ratio/mean": 0.99709050655365, |
| "sampling/importance_sampling_ratio/min": 0.6305613279342651, |
| "sampling/sampling_logp_difference/max": 0.4242278516292572, |
| "sampling/sampling_logp_difference/mean": 0.0008203186574974097, |
| "step": 450, |
| "step_time": 4.895320072211325 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 109.7, |
| "completions/max_terminated_length": 109.7, |
| "completions/mean_length": 49.6859375, |
| "completions/mean_terminated_length": 49.6859375, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.008923626859905198, |
| "epoch": 1.1886304909560723, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.369140625, |
| "learning_rate": 9.954100000000002e-06, |
| "loss": -0.0012, |
| "num_tokens": 25467840.0, |
| "reward": 0.27574220299720764, |
| "reward_std": 0.3561663553118706, |
| "rewards/reward_accuracy/mean": 0.17578125, |
| "rewards/reward_accuracy/std": 0.35613545030355453, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.5703716278076172, |
| "sampling/importance_sampling_ratio/mean": 0.9965644836425781, |
| "sampling/importance_sampling_ratio/min": 0.5899728700518608, |
| "sampling/sampling_logp_difference/max": 0.5821703374385834, |
| "sampling/sampling_logp_difference/mean": 0.0012180614110548049, |
| "step": 460, |
| "step_time": 5.456775700231082 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.460280327592045e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.460280327592045e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 51.696875, |
| "completions/mean_terminated_length": 51.696875, |
| "completions/min_length": 26.9, |
| "completions/min_terminated_length": 26.9, |
| "entropy": 0.0074561212779372, |
| "epoch": 1.2144702842377262, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 1.2109375, |
| "learning_rate": 9.953100000000001e-06, |
| "loss": 0.0014, |
| "num_tokens": 26023380.0, |
| "reward": 0.34375002086162565, |
| "reward_std": 0.41014502495527266, |
| "rewards/reward_accuracy/mean": 0.24375, |
| "rewards/reward_accuracy/std": 0.41014502197504044, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.47359801530838, |
| "sampling/importance_sampling_ratio/mean": 1.003223365545273, |
| "sampling/importance_sampling_ratio/min": 0.6739776849746704, |
| "sampling/sampling_logp_difference/max": 0.45461962223052976, |
| "sampling/sampling_logp_difference/mean": 0.0008782176155364141, |
| "step": 470, |
| "step_time": 4.948963294457644 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.3, |
| "completions/max_terminated_length": 73.3, |
| "completions/mean_length": 53.21796875, |
| "completions/mean_terminated_length": 53.21796875, |
| "completions/min_length": 27.5, |
| "completions/min_terminated_length": 27.5, |
| "entropy": 0.009096419914567378, |
| "epoch": 1.2403100775193798, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 1.0078125, |
| "learning_rate": 9.9521e-06, |
| "loss": 0.0031, |
| "num_tokens": 26584371.0, |
| "reward": 0.3171875193715096, |
| "reward_std": 0.3827620230615139, |
| "rewards/reward_accuracy/mean": 0.2171875, |
| "rewards/reward_accuracy/std": 0.3827620230615139, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5909615635871888, |
| "sampling/importance_sampling_ratio/mean": 0.99535653591156, |
| "sampling/importance_sampling_ratio/min": 0.6510380163788796, |
| "sampling/sampling_logp_difference/max": 0.6225070595741272, |
| "sampling/sampling_logp_difference/mean": 0.0012263015523785725, |
| "step": 480, |
| "step_time": 5.03073446888011 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 76.6, |
| "completions/max_terminated_length": 76.6, |
| "completions/mean_length": 50.753125, |
| "completions/mean_terminated_length": 50.753125, |
| "completions/min_length": 28.8, |
| "completions/min_terminated_length": 28.8, |
| "entropy": 0.007166365180455614, |
| "epoch": 1.2661498708010335, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 1.2578125, |
| "learning_rate": 9.951100000000002e-06, |
| "loss": 0.0092, |
| "num_tokens": 27136479.0, |
| "reward": 0.28984377086162566, |
| "reward_std": 0.38223251402378083, |
| "rewards/reward_accuracy/mean": 0.18984375, |
| "rewards/reward_accuracy/std": 0.38223251402378083, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5218806266784668, |
| "sampling/importance_sampling_ratio/mean": 1.0051551342010498, |
| "sampling/importance_sampling_ratio/min": 0.6391298770904541, |
| "sampling/sampling_logp_difference/max": 0.5423847794532776, |
| "sampling/sampling_logp_difference/mean": 0.0009278648518375121, |
| "step": 490, |
| "step_time": 5.0575280277291315 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.8, |
| "completions/max_terminated_length": 71.8, |
| "completions/mean_length": 48.659375, |
| "completions/mean_terminated_length": 48.659375, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.007065431242517661, |
| "epoch": 1.2919896640826873, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.97265625, |
| "learning_rate": 9.9501e-06, |
| "loss": -0.0014, |
| "num_tokens": 27681483.0, |
| "reward": 0.3742187738418579, |
| "reward_std": 0.4437094271183014, |
| "rewards/reward_accuracy/mean": 0.27421875, |
| "rewards/reward_accuracy/std": 0.44370942413806913, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.518882405757904, |
| "sampling/importance_sampling_ratio/mean": 0.999101185798645, |
| "sampling/importance_sampling_ratio/min": 0.5955654293298721, |
| "sampling/sampling_logp_difference/max": 0.46556512713432313, |
| "sampling/sampling_logp_difference/mean": 0.000992022972786799, |
| "step": 500, |
| "step_time": 4.973484661174007 |
| }, |
| { |
| "clip_ratio/high_max": 7.123829564079643e-05, |
| "clip_ratio/high_mean": 3.5619147820398214e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.5619147820398214e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 94.6, |
| "completions/max_terminated_length": 94.6, |
| "completions/mean_length": 50.6515625, |
| "completions/mean_terminated_length": 50.6515625, |
| "completions/min_length": 25.8, |
| "completions/min_terminated_length": 25.8, |
| "entropy": 0.0057475918903946875, |
| "epoch": 1.3178294573643412, |
| "frac_reward_zero_std": 0.91875, |
| "grad_norm": 1.1640625, |
| "learning_rate": 9.949100000000001e-06, |
| "loss": 0.0031, |
| "num_tokens": 28231373.0, |
| "reward": 0.3398437723517418, |
| "reward_std": 0.4180413454771042, |
| "rewards/reward_accuracy/mean": 0.23984375, |
| "rewards/reward_accuracy/std": 0.41804134249687197, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5994762539863587, |
| "sampling/importance_sampling_ratio/mean": 1.001958680152893, |
| "sampling/importance_sampling_ratio/min": 0.633919720351696, |
| "sampling/sampling_logp_difference/max": 0.7151550352573395, |
| "sampling/sampling_logp_difference/mean": 0.0008901450491975993, |
| "step": 510, |
| "step_time": 5.286917511024512 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 76.3, |
| "completions/max_terminated_length": 76.3, |
| "completions/mean_length": 49.7625, |
| "completions/mean_terminated_length": 49.7625, |
| "completions/min_length": 24.6, |
| "completions/min_terminated_length": 24.6, |
| "entropy": 0.004566283169333473, |
| "epoch": 1.3436692506459949, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.3359375, |
| "learning_rate": 9.9481e-06, |
| "loss": 0.0016, |
| "num_tokens": 28780093.0, |
| "reward": 0.3828125223517418, |
| "reward_std": 0.43982950448989866, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.43982950448989866, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4449692845344544, |
| "sampling/importance_sampling_ratio/mean": 1.0004740953445435, |
| "sampling/importance_sampling_ratio/min": 0.5853417366743088, |
| "sampling/sampling_logp_difference/max": 0.6157394766807556, |
| "sampling/sampling_logp_difference/mean": 0.0005028381056035869, |
| "step": 520, |
| "step_time": 5.00728235852439 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.4, |
| "completions/max_terminated_length": 74.4, |
| "completions/mean_length": 48.79140625, |
| "completions/mean_terminated_length": 48.79140625, |
| "completions/min_length": 27.6, |
| "completions/min_terminated_length": 27.6, |
| "entropy": 0.004568320088947075, |
| "epoch": 1.3695090439276485, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.185546875, |
| "learning_rate": 9.9471e-06, |
| "loss": 0.0001, |
| "num_tokens": 29322754.0, |
| "reward": 0.3406250193715096, |
| "reward_std": 0.4137811928987503, |
| "rewards/reward_accuracy/mean": 0.240625, |
| "rewards/reward_accuracy/std": 0.4137811928987503, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4529842138290405, |
| "sampling/importance_sampling_ratio/mean": 1.0002224147319794, |
| "sampling/importance_sampling_ratio/min": 0.6508453249931335, |
| "sampling/sampling_logp_difference/max": 0.4745017230510712, |
| "sampling/sampling_logp_difference/mean": 0.0006546025455463677, |
| "step": 530, |
| "step_time": 4.99087378543336 |
| }, |
| { |
| "clip_ratio/high_max": 9.566595836076885e-05, |
| "clip_ratio/high_mean": 4.7832979180384425e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.7832979180384425e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 79.6, |
| "completions/max_terminated_length": 79.6, |
| "completions/mean_length": 51.58515625, |
| "completions/mean_terminated_length": 51.58515625, |
| "completions/min_length": 26.2, |
| "completions/min_terminated_length": 26.2, |
| "entropy": 0.004724874184830696, |
| "epoch": 1.3953488372093024, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.946100000000001e-06, |
| "loss": 0.0011, |
| "num_tokens": 29877495.0, |
| "reward": 0.4164062738418579, |
| "reward_std": 0.45978312492370604, |
| "rewards/reward_accuracy/mean": 0.31640625, |
| "rewards/reward_accuracy/std": 0.45978312492370604, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5026076316833497, |
| "sampling/importance_sampling_ratio/mean": 0.9967418015003204, |
| "sampling/importance_sampling_ratio/min": 0.5300931870937348, |
| "sampling/sampling_logp_difference/max": 0.6870249032974243, |
| "sampling/sampling_logp_difference/mean": 0.0008436903262918349, |
| "step": 540, |
| "step_time": 5.0624481728766115 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.3, |
| "completions/max_terminated_length": 73.3, |
| "completions/mean_length": 50.9671875, |
| "completions/mean_terminated_length": 50.9671875, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.004270565678598359, |
| "epoch": 1.421188630490956, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9451e-06, |
| "loss": 0.0008, |
| "num_tokens": 30429373.0, |
| "reward": 0.30468752086162565, |
| "reward_std": 0.39117962718009947, |
| "rewards/reward_accuracy/mean": 0.2046875, |
| "rewards/reward_accuracy/std": 0.39117962718009947, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.587038016319275, |
| "sampling/importance_sampling_ratio/mean": 0.9992846667766571, |
| "sampling/importance_sampling_ratio/min": 0.651472982764244, |
| "sampling/sampling_logp_difference/max": 0.5502383947372437, |
| "sampling/sampling_logp_difference/mean": 0.0008272819832200184, |
| "step": 550, |
| "step_time": 5.060468366672285 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.1, |
| "completions/max_terminated_length": 72.1, |
| "completions/mean_length": 49.85234375, |
| "completions/mean_terminated_length": 49.85234375, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.004139559478244337, |
| "epoch": 1.4470284237726099, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 1.5859375, |
| "learning_rate": 9.9441e-06, |
| "loss": -0.002, |
| "num_tokens": 30976992.0, |
| "reward": 0.3687500238418579, |
| "reward_std": 0.4345968633890152, |
| "rewards/reward_accuracy/mean": 0.26875, |
| "rewards/reward_accuracy/std": 0.4345968633890152, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6781012535095214, |
| "sampling/importance_sampling_ratio/mean": 1.0005912721157073, |
| "sampling/importance_sampling_ratio/min": 0.5941987067461014, |
| "sampling/sampling_logp_difference/max": 0.7543541431427002, |
| "sampling/sampling_logp_difference/mean": 0.0007979679765412584, |
| "step": 560, |
| "step_time": 5.010671874531544 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.9, |
| "completions/max_terminated_length": 71.9, |
| "completions/mean_length": 50.30625, |
| "completions/mean_terminated_length": 50.30625, |
| "completions/min_length": 28.1, |
| "completions/min_terminated_length": 28.1, |
| "entropy": 0.004270606058344129, |
| "epoch": 1.4728682170542635, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.943100000000001e-06, |
| "loss": 0.0029, |
| "num_tokens": 31525816.0, |
| "reward": 0.3609375238418579, |
| "reward_std": 0.43490211963653563, |
| "rewards/reward_accuracy/mean": 0.2609375, |
| "rewards/reward_accuracy/std": 0.4349021166563034, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5483613848686217, |
| "sampling/importance_sampling_ratio/mean": 0.9988287806510925, |
| "sampling/importance_sampling_ratio/min": 0.7706151485443116, |
| "sampling/sampling_logp_difference/max": 0.5030603908002377, |
| "sampling/sampling_logp_difference/mean": 0.0006874992242956069, |
| "step": 570, |
| "step_time": 4.850441494234838 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 91.3, |
| "completions/max_terminated_length": 91.3, |
| "completions/mean_length": 49.16484375, |
| "completions/mean_terminated_length": 49.16484375, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.005170105910474376, |
| "epoch": 1.4987080103359174, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.942100000000001e-06, |
| "loss": -0.002, |
| "num_tokens": 32070067.0, |
| "reward": 0.36093752086162567, |
| "reward_std": 0.426634755730629, |
| "rewards/reward_accuracy/mean": 0.2609375, |
| "rewards/reward_accuracy/std": 0.426634755730629, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3277226209640502, |
| "sampling/importance_sampling_ratio/mean": 0.9973502337932587, |
| "sampling/importance_sampling_ratio/min": 0.6891504764556885, |
| "sampling/sampling_logp_difference/max": 0.3815302750095725, |
| "sampling/sampling_logp_difference/mean": 0.0006895444366818992, |
| "step": 580, |
| "step_time": 5.169872325286269 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 4.7017035831231625e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.7017035831231625e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 81.2, |
| "completions/max_terminated_length": 81.2, |
| "completions/mean_length": 51.49375, |
| "completions/mean_terminated_length": 51.49375, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.005989666550885886, |
| "epoch": 1.524547803617571, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.941100000000002e-06, |
| "loss": -0.0001, |
| "num_tokens": 32623155.0, |
| "reward": 0.3703125223517418, |
| "reward_std": 0.43225965797901156, |
| "rewards/reward_accuracy/mean": 0.2703125, |
| "rewards/reward_accuracy/std": 0.4322596549987793, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.705933403968811, |
| "sampling/importance_sampling_ratio/mean": 1.0014257848262786, |
| "sampling/importance_sampling_ratio/min": 0.6864098310470581, |
| "sampling/sampling_logp_difference/max": 0.5644432842731476, |
| "sampling/sampling_logp_difference/mean": 0.0009663421020377428, |
| "step": 590, |
| "step_time": 5.0122690113727 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.4, |
| "completions/max_terminated_length": 70.4, |
| "completions/mean_length": 47.0171875, |
| "completions/mean_terminated_length": 47.0171875, |
| "completions/min_length": 24.2, |
| "completions/min_terminated_length": 24.2, |
| "entropy": 0.00489957259305811, |
| "epoch": 1.550387596899225, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9401e-06, |
| "loss": 0.0018, |
| "num_tokens": 33159361.0, |
| "reward": 0.3773437708616257, |
| "reward_std": 0.4363934338092804, |
| "rewards/reward_accuracy/mean": 0.27734375, |
| "rewards/reward_accuracy/std": 0.4363934338092804, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4506001949310303, |
| "sampling/importance_sampling_ratio/mean": 0.9965661525726318, |
| "sampling/importance_sampling_ratio/min": 0.5474170207977295, |
| "sampling/sampling_logp_difference/max": 0.6956219136714935, |
| "sampling/sampling_logp_difference/mean": 0.0009393148910021409, |
| "step": 600, |
| "step_time": 4.871894678263925 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.3, |
| "completions/max_terminated_length": 71.3, |
| "completions/mean_length": 49.834375, |
| "completions/mean_terminated_length": 49.834375, |
| "completions/min_length": 28.9, |
| "completions/min_terminated_length": 28.9, |
| "entropy": 0.004814997518042219, |
| "epoch": 1.5762273901808785, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 1.3125, |
| "learning_rate": 9.939100000000001e-06, |
| "loss": -0.0011, |
| "num_tokens": 33707613.0, |
| "reward": 0.3171875223517418, |
| "reward_std": 0.4023925080895424, |
| "rewards/reward_accuracy/mean": 0.2171875, |
| "rewards/reward_accuracy/std": 0.4023925080895424, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4736223101615906, |
| "sampling/importance_sampling_ratio/mean": 1.0025293171405791, |
| "sampling/importance_sampling_ratio/min": 0.5906230822205544, |
| "sampling/sampling_logp_difference/max": 0.6017002820968628, |
| "sampling/sampling_logp_difference/mean": 0.0009015583724249155, |
| "step": 610, |
| "step_time": 4.924097525980324 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 51.25859375, |
| "completions/mean_terminated_length": 51.25859375, |
| "completions/min_length": 26.7, |
| "completions/min_terminated_length": 26.7, |
| "entropy": 0.005248205434691044, |
| "epoch": 1.6020671834625322, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 1.84375, |
| "learning_rate": 9.9381e-06, |
| "loss": -0.0055, |
| "num_tokens": 34261888.0, |
| "reward": 0.34765626937150956, |
| "reward_std": 0.41243806183338166, |
| "rewards/reward_accuracy/mean": 0.24765625, |
| "rewards/reward_accuracy/std": 0.41243806183338166, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6348495602607727, |
| "sampling/importance_sampling_ratio/mean": 1.003392642736435, |
| "sampling/importance_sampling_ratio/min": 0.6136953711509705, |
| "sampling/sampling_logp_difference/max": 0.6362825155258178, |
| "sampling/sampling_logp_difference/mean": 0.0009955753339454532, |
| "step": 620, |
| "step_time": 4.994224854256027 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.1, |
| "completions/max_terminated_length": 72.1, |
| "completions/mean_length": 50.1671875, |
| "completions/mean_terminated_length": 50.1671875, |
| "completions/min_length": 26.8, |
| "completions/min_terminated_length": 26.8, |
| "entropy": 0.004563965863053454, |
| "epoch": 1.627906976744186, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 1.125, |
| "learning_rate": 9.9371e-06, |
| "loss": -0.0038, |
| "num_tokens": 34811990.0, |
| "reward": 0.3218750223517418, |
| "reward_std": 0.399589104950428, |
| "rewards/reward_accuracy/mean": 0.221875, |
| "rewards/reward_accuracy/std": 0.39958910197019576, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4449053406715393, |
| "sampling/importance_sampling_ratio/mean": 0.9982593476772308, |
| "sampling/importance_sampling_ratio/min": 0.5630527019500733, |
| "sampling/sampling_logp_difference/max": 0.6468969106674194, |
| "sampling/sampling_logp_difference/mean": 0.0007250470865983516, |
| "step": 630, |
| "step_time": 4.995986625179649 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 50.78828125, |
| "completions/mean_terminated_length": 50.78828125, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.005075658496571123, |
| "epoch": 1.65374677002584, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 2.09375, |
| "learning_rate": 9.936100000000001e-06, |
| "loss": 0.0071, |
| "num_tokens": 35363463.0, |
| "reward": 0.3476562723517418, |
| "reward_std": 0.40936014503240586, |
| "rewards/reward_accuracy/mean": 0.24765625, |
| "rewards/reward_accuracy/std": 0.40936014503240586, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4420273661613465, |
| "sampling/importance_sampling_ratio/mean": 1.0013975381851197, |
| "sampling/importance_sampling_ratio/min": 0.6976974219083786, |
| "sampling/sampling_logp_difference/max": 0.48383174538612367, |
| "sampling/sampling_logp_difference/mean": 0.0008290811092592776, |
| "step": 640, |
| "step_time": 4.916350444639102 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.8, |
| "completions/max_terminated_length": 70.8, |
| "completions/mean_length": 48.3515625, |
| "completions/mean_terminated_length": 48.3515625, |
| "completions/min_length": 24.7, |
| "completions/min_terminated_length": 24.7, |
| "entropy": 0.003946251096567721, |
| "epoch": 1.6795865633074936, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.97265625, |
| "learning_rate": 9.9351e-06, |
| "loss": 0.0053, |
| "num_tokens": 35906945.0, |
| "reward": 0.36484377160668374, |
| "reward_std": 0.40230888724327085, |
| "rewards/reward_accuracy/mean": 0.26484375, |
| "rewards/reward_accuracy/std": 0.40230888724327085, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4134734511375426, |
| "sampling/importance_sampling_ratio/mean": 1.00091295838356, |
| "sampling/importance_sampling_ratio/min": 0.6098767161369324, |
| "sampling/sampling_logp_difference/max": 0.5155020594596863, |
| "sampling/sampling_logp_difference/mean": 0.0005970991813228465, |
| "step": 650, |
| "step_time": 4.914593392447569 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.1, |
| "completions/max_terminated_length": 71.1, |
| "completions/mean_length": 48.68203125, |
| "completions/mean_terminated_length": 48.68203125, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.005265657445852412, |
| "epoch": 1.7054263565891472, |
| "frac_reward_zero_std": 0.91875, |
| "grad_norm": 0.5390625, |
| "learning_rate": 9.9341e-06, |
| "loss": -0.003, |
| "num_tokens": 36449898.0, |
| "reward": 0.32109377086162566, |
| "reward_std": 0.4015688106417656, |
| "rewards/reward_accuracy/mean": 0.22109375, |
| "rewards/reward_accuracy/std": 0.4015688106417656, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4113248705863952, |
| "sampling/importance_sampling_ratio/mean": 1.0041666328907013, |
| "sampling/importance_sampling_ratio/min": 0.6386713862419129, |
| "sampling/sampling_logp_difference/max": 0.5526663899421692, |
| "sampling/sampling_logp_difference/mean": 0.0008046192408073694, |
| "step": 660, |
| "step_time": 4.9653470883145925 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 120.1, |
| "completions/max_terminated_length": 120.1, |
| "completions/mean_length": 51.3140625, |
| "completions/mean_terminated_length": 51.3140625, |
| "completions/min_length": 25.8, |
| "completions/min_terminated_length": 25.8, |
| "entropy": 0.006548115975601831, |
| "epoch": 1.731266149870801, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.933100000000002e-06, |
| "loss": -0.002, |
| "num_tokens": 37001852.0, |
| "reward": 0.3210937723517418, |
| "reward_std": 0.40834735333919525, |
| "rewards/reward_accuracy/mean": 0.22109375, |
| "rewards/reward_accuracy/std": 0.40834735333919525, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5027079701423645, |
| "sampling/importance_sampling_ratio/mean": 0.9997454047203064, |
| "sampling/importance_sampling_ratio/min": 0.7014189481735229, |
| "sampling/sampling_logp_difference/max": 0.45845555067062377, |
| "sampling/sampling_logp_difference/mean": 0.0008648375238408335, |
| "step": 670, |
| "step_time": 5.6162318653659895 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.9, |
| "completions/max_terminated_length": 71.9, |
| "completions/mean_length": 48.8859375, |
| "completions/mean_terminated_length": 48.8859375, |
| "completions/min_length": 28.2, |
| "completions/min_terminated_length": 28.2, |
| "entropy": 0.005832143905718112, |
| "epoch": 1.757105943152455, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.80078125, |
| "learning_rate": 9.932100000000001e-06, |
| "loss": -0.0013, |
| "num_tokens": 37546226.0, |
| "reward": 0.34218752235174177, |
| "reward_std": 0.4177682489156723, |
| "rewards/reward_accuracy/mean": 0.2421875, |
| "rewards/reward_accuracy/std": 0.4177682489156723, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3385825395584106, |
| "sampling/importance_sampling_ratio/mean": 0.9974124372005463, |
| "sampling/importance_sampling_ratio/min": 0.5953316509723663, |
| "sampling/sampling_logp_difference/max": 0.6052889108657837, |
| "sampling/sampling_logp_difference/mean": 0.000901306263403967, |
| "step": 680, |
| "step_time": 4.981575359357521 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 49.7515625, |
| "completions/mean_terminated_length": 49.7515625, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.005523035162332235, |
| "epoch": 1.7829457364341086, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 2.234375, |
| "learning_rate": 9.9311e-06, |
| "loss": 0.0008, |
| "num_tokens": 38093044.0, |
| "reward": 0.3054687693715096, |
| "reward_std": 0.37578703463077545, |
| "rewards/reward_accuracy/mean": 0.20546875, |
| "rewards/reward_accuracy/std": 0.375787028670311, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3629189729690552, |
| "sampling/importance_sampling_ratio/mean": 0.994739580154419, |
| "sampling/importance_sampling_ratio/min": 0.5167732790112496, |
| "sampling/sampling_logp_difference/max": 0.7097882837057113, |
| "sampling/sampling_logp_difference/mean": 0.001015731597726699, |
| "step": 690, |
| "step_time": 4.911289026914164 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 93.9, |
| "completions/max_terminated_length": 93.9, |
| "completions/mean_length": 52.2328125, |
| "completions/mean_terminated_length": 52.2328125, |
| "completions/min_length": 23.7, |
| "completions/min_terminated_length": 23.7, |
| "entropy": 0.006199735609698109, |
| "epoch": 1.8087855297157622, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.79296875, |
| "learning_rate": 9.9301e-06, |
| "loss": 0.0034, |
| "num_tokens": 38650102.0, |
| "reward": 0.2789062663912773, |
| "reward_std": 0.3607895582914352, |
| "rewards/reward_accuracy/mean": 0.17890625, |
| "rewards/reward_accuracy/std": 0.3607895582914352, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5490392088890075, |
| "sampling/importance_sampling_ratio/mean": 1.0021339178085327, |
| "sampling/importance_sampling_ratio/min": 0.6575803846120835, |
| "sampling/sampling_logp_difference/max": 0.5140547037124634, |
| "sampling/sampling_logp_difference/mean": 0.0007813677075318992, |
| "step": 700, |
| "step_time": 5.34737250553444 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 49.3546875, |
| "completions/mean_terminated_length": 49.3546875, |
| "completions/min_length": 26.1, |
| "completions/min_terminated_length": 26.1, |
| "entropy": 0.005531836655427469, |
| "epoch": 1.8346253229974159, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.929100000000001e-06, |
| "loss": 0.0053, |
| "num_tokens": 39195604.0, |
| "reward": 0.3539062708616257, |
| "reward_std": 0.4250497162342072, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4250497162342072, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5420080542564392, |
| "sampling/importance_sampling_ratio/mean": 1.001390928030014, |
| "sampling/importance_sampling_ratio/min": 0.7318558424711228, |
| "sampling/sampling_logp_difference/max": 0.47918625473976134, |
| "sampling/sampling_logp_difference/mean": 0.000776525036781095, |
| "step": 710, |
| "step_time": 4.962767287250609 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 49.03984375, |
| "completions/mean_terminated_length": 49.03984375, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.005406915253843181, |
| "epoch": 1.8604651162790697, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.9765625, |
| "learning_rate": 9.9281e-06, |
| "loss": -0.0024, |
| "num_tokens": 39739831.0, |
| "reward": 0.37890627086162565, |
| "reward_std": 0.42426152527332306, |
| "rewards/reward_accuracy/mean": 0.27890625, |
| "rewards/reward_accuracy/std": 0.42426152527332306, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5376534342765809, |
| "sampling/importance_sampling_ratio/mean": 1.0025013029575347, |
| "sampling/importance_sampling_ratio/min": 0.6273042261600494, |
| "sampling/sampling_logp_difference/max": 0.6381498456001282, |
| "sampling/sampling_logp_difference/mean": 0.0008977889243396931, |
| "step": 720, |
| "step_time": 4.881878258194774 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 75.9, |
| "completions/max_terminated_length": 75.9, |
| "completions/mean_length": 51.2796875, |
| "completions/mean_terminated_length": 51.2796875, |
| "completions/min_length": 29.1, |
| "completions/min_terminated_length": 29.1, |
| "entropy": 0.005592045963203418, |
| "epoch": 1.8863049095607236, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9271e-06, |
| "loss": -0.0014, |
| "num_tokens": 40293757.0, |
| "reward": 0.31875001788139345, |
| "reward_std": 0.38975515216588974, |
| "rewards/reward_accuracy/mean": 0.21875, |
| "rewards/reward_accuracy/std": 0.38975515216588974, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2487102508544923, |
| "sampling/importance_sampling_ratio/mean": 0.9949997544288636, |
| "sampling/importance_sampling_ratio/min": 0.6795156002044678, |
| "sampling/sampling_logp_difference/max": 0.4270708441734314, |
| "sampling/sampling_logp_difference/mean": 0.00059806551435031, |
| "step": 730, |
| "step_time": 5.083559083682485 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.5, |
| "completions/max_terminated_length": 72.5, |
| "completions/mean_length": 49.32578125, |
| "completions/mean_terminated_length": 49.32578125, |
| "completions/min_length": 24.4, |
| "completions/min_terminated_length": 24.4, |
| "entropy": 0.004020834702532739, |
| "epoch": 1.9121447028423773, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.926100000000001e-06, |
| "loss": -0.0006, |
| "num_tokens": 40840070.0, |
| "reward": 0.349960957467556, |
| "reward_std": 0.41404010355472565, |
| "rewards/reward_accuracy/mean": 0.25, |
| "rewards/reward_accuracy/std": 0.4140223443508148, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.4729785919189453, |
| "sampling/importance_sampling_ratio/mean": 1.002078151702881, |
| "sampling/importance_sampling_ratio/min": 0.7183513581752777, |
| "sampling/sampling_logp_difference/max": 0.5612375795841217, |
| "sampling/sampling_logp_difference/mean": 0.0006096532102674246, |
| "step": 740, |
| "step_time": 4.955068456544541 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 104.2, |
| "completions/max_terminated_length": 104.2, |
| "completions/mean_length": 50.95703125, |
| "completions/mean_terminated_length": 50.95703125, |
| "completions/min_length": 28.4, |
| "completions/min_terminated_length": 28.4, |
| "entropy": 0.004746937126037664, |
| "epoch": 1.937984496124031, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 1.2109375, |
| "learning_rate": 9.925100000000001e-06, |
| "loss": 0.0005, |
| "num_tokens": 41392503.0, |
| "reward": 0.4054687723517418, |
| "reward_std": 0.43919114023447037, |
| "rewards/reward_accuracy/mean": 0.30546875, |
| "rewards/reward_accuracy/std": 0.43919113725423814, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.359642744064331, |
| "sampling/importance_sampling_ratio/mean": 1.0001107513904572, |
| "sampling/importance_sampling_ratio/min": 0.7032971888780594, |
| "sampling/sampling_logp_difference/max": 0.372553151845932, |
| "sampling/sampling_logp_difference/mean": 0.0005954385080258362, |
| "step": 750, |
| "step_time": 5.42899293450173 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.1, |
| "completions/max_terminated_length": 72.1, |
| "completions/mean_length": 49.20234375, |
| "completions/mean_terminated_length": 49.20234375, |
| "completions/min_length": 27.8, |
| "completions/min_terminated_length": 27.8, |
| "entropy": 0.0052071080452151365, |
| "epoch": 1.9638242894056848, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 1.40625, |
| "learning_rate": 9.9241e-06, |
| "loss": -0.0049, |
| "num_tokens": 41937266.0, |
| "reward": 0.3265625208616257, |
| "reward_std": 0.4094372928142548, |
| "rewards/reward_accuracy/mean": 0.2265625, |
| "rewards/reward_accuracy/std": 0.4094372928142548, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4685480833053588, |
| "sampling/importance_sampling_ratio/mean": 1.0023519039154052, |
| "sampling/importance_sampling_ratio/min": 0.6667828857898712, |
| "sampling/sampling_logp_difference/max": 0.48669140338897704, |
| "sampling/sampling_logp_difference/mean": 0.0010033050057245418, |
| "step": 760, |
| "step_time": 4.922246447415091 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.9, |
| "completions/max_terminated_length": 72.9, |
| "completions/mean_length": 51.275, |
| "completions/mean_terminated_length": 51.275, |
| "completions/min_length": 28.4, |
| "completions/min_terminated_length": 28.4, |
| "entropy": 0.004649722462272621, |
| "epoch": 1.9896640826873386, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.923100000000002e-06, |
| "loss": 0.0014, |
| "num_tokens": 42490514.0, |
| "reward": 0.3445312723517418, |
| "reward_std": 0.41868719309568403, |
| "rewards/reward_accuracy/mean": 0.24453125, |
| "rewards/reward_accuracy/std": 0.4186871901154518, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3083459258079528, |
| "sampling/importance_sampling_ratio/mean": 1.0007959306240082, |
| "sampling/importance_sampling_ratio/min": 0.6328578025102616, |
| "sampling/sampling_logp_difference/max": 0.4959941267967224, |
| "sampling/sampling_logp_difference/mean": 0.000599285590578802, |
| "step": 770, |
| "step_time": 4.951897845813074 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 130.9, |
| "completions/max_terminated_length": 130.9, |
| "completions/mean_length": 50.83984375, |
| "completions/mean_terminated_length": 50.83984375, |
| "completions/min_length": 27.7, |
| "completions/min_terminated_length": 27.7, |
| "entropy": 0.004862227219564375, |
| "epoch": 2.0155038759689923, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.922100000000001e-06, |
| "loss": 0.0011, |
| "num_tokens": 43041917.0, |
| "reward": 0.3445312723517418, |
| "reward_std": 0.4205166459083557, |
| "rewards/reward_accuracy/mean": 0.24453125, |
| "rewards/reward_accuracy/std": 0.4205166459083557, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.355384314060211, |
| "sampling/importance_sampling_ratio/mean": 1.0005921304225922, |
| "sampling/importance_sampling_ratio/min": 0.5565952569246292, |
| "sampling/sampling_logp_difference/max": 0.6045052528381347, |
| "sampling/sampling_logp_difference/mean": 0.0008310028031701222, |
| "step": 780, |
| "step_time": 5.81024782792665 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 50.57265625, |
| "completions/mean_terminated_length": 50.57265625, |
| "completions/min_length": 29.2, |
| "completions/min_terminated_length": 29.2, |
| "entropy": 0.0034937210055431935, |
| "epoch": 2.041343669250646, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 0.462890625, |
| "learning_rate": 9.9211e-06, |
| "loss": 0.0028, |
| "num_tokens": 43591810.0, |
| "reward": 0.30078126639127734, |
| "reward_std": 0.3889342963695526, |
| "rewards/reward_accuracy/mean": 0.20078125, |
| "rewards/reward_accuracy/std": 0.3889342963695526, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3001936674118042, |
| "sampling/importance_sampling_ratio/mean": 0.9966077208518982, |
| "sampling/importance_sampling_ratio/min": 0.6751231610774994, |
| "sampling/sampling_logp_difference/max": 0.4490883946418762, |
| "sampling/sampling_logp_difference/mean": 0.0005357342117349618, |
| "step": 790, |
| "step_time": 4.898826712369919 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 48.78671875, |
| "completions/mean_terminated_length": 48.78671875, |
| "completions/min_length": 25.8, |
| "completions/min_terminated_length": 25.8, |
| "entropy": 0.0036701811619423096, |
| "epoch": 2.0671834625322996, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 2.09375, |
| "learning_rate": 9.9201e-06, |
| "loss": 0.0016, |
| "num_tokens": 44137849.0, |
| "reward": 0.3351562723517418, |
| "reward_std": 0.4086653530597687, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.4086653530597687, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6247925281524658, |
| "sampling/importance_sampling_ratio/mean": 1.0063852429389955, |
| "sampling/importance_sampling_ratio/min": 0.6922838568687439, |
| "sampling/sampling_logp_difference/max": 0.5680738091468811, |
| "sampling/sampling_logp_difference/mean": 0.0006539385438372846, |
| "step": 800, |
| "step_time": 4.966615257156081 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.8168604583479464e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.8168604583479464e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.9, |
| "completions/max_terminated_length": 71.9, |
| "completions/mean_length": 48.7765625, |
| "completions/mean_terminated_length": 48.7765625, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.004219953557367262, |
| "epoch": 2.0930232558139537, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 1.3359375, |
| "learning_rate": 9.9191e-06, |
| "loss": -0.0008, |
| "num_tokens": 44682635.0, |
| "reward": 0.34453127086162566, |
| "reward_std": 0.41236633211374285, |
| "rewards/reward_accuracy/mean": 0.24453125, |
| "rewards/reward_accuracy/std": 0.41236633211374285, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2463948369026183, |
| "sampling/importance_sampling_ratio/mean": 0.9972876310348511, |
| "sampling/importance_sampling_ratio/min": 0.6533996075391769, |
| "sampling/sampling_logp_difference/max": 0.4773648500442505, |
| "sampling/sampling_logp_difference/mean": 0.000641372000973206, |
| "step": 810, |
| "step_time": 5.038309122831561 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.5, |
| "completions/max_terminated_length": 71.5, |
| "completions/mean_length": 50.23828125, |
| "completions/mean_terminated_length": 50.23828125, |
| "completions/min_length": 28.3, |
| "completions/min_terminated_length": 28.3, |
| "entropy": 0.0036592685602954587, |
| "epoch": 2.1188630490956073, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.921875, |
| "learning_rate": 9.9181e-06, |
| "loss": -0.0022, |
| "num_tokens": 45233980.0, |
| "reward": 0.3367187693715096, |
| "reward_std": 0.40002884417772294, |
| "rewards/reward_accuracy/mean": 0.23671875, |
| "rewards/reward_accuracy/std": 0.40002884417772294, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2568351745605468, |
| "sampling/importance_sampling_ratio/mean": 0.9993258237838745, |
| "sampling/importance_sampling_ratio/min": 0.7089363902807235, |
| "sampling/sampling_logp_difference/max": 0.4356670379638672, |
| "sampling/sampling_logp_difference/mean": 0.00038460935320472345, |
| "step": 820, |
| "step_time": 4.8515168277546765 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 77.4, |
| "completions/max_terminated_length": 77.4, |
| "completions/mean_length": 49.01171875, |
| "completions/mean_terminated_length": 49.01171875, |
| "completions/min_length": 27.5, |
| "completions/min_terminated_length": 27.5, |
| "entropy": 0.005215529868655722, |
| "epoch": 2.144702842377261, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9171e-06, |
| "loss": -0.0005, |
| "num_tokens": 45779043.0, |
| "reward": 0.3703125238418579, |
| "reward_std": 0.431861937046051, |
| "rewards/reward_accuracy/mean": 0.2703125, |
| "rewards/reward_accuracy/std": 0.431861937046051, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5551151514053345, |
| "sampling/importance_sampling_ratio/mean": 0.9947058022022247, |
| "sampling/importance_sampling_ratio/min": 0.6013443797826767, |
| "sampling/sampling_logp_difference/max": 0.619977805018425, |
| "sampling/sampling_logp_difference/mean": 0.000956929786480032, |
| "step": 830, |
| "step_time": 5.017891668272204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.1, |
| "completions/max_terminated_length": 72.1, |
| "completions/mean_length": 50.15390625, |
| "completions/mean_terminated_length": 50.15390625, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.004304639185374981, |
| "epoch": 2.1705426356589146, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.916100000000002e-06, |
| "loss": -0.0038, |
| "num_tokens": 46327224.0, |
| "reward": 0.37500002384185793, |
| "reward_std": 0.43294604420661925, |
| "rewards/reward_accuracy/mean": 0.275, |
| "rewards/reward_accuracy/std": 0.43294604420661925, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.507475233078003, |
| "sampling/importance_sampling_ratio/mean": 1.001672661304474, |
| "sampling/importance_sampling_ratio/min": 0.6584469556808472, |
| "sampling/sampling_logp_difference/max": 0.5379517555236817, |
| "sampling/sampling_logp_difference/mean": 0.000783838582719909, |
| "step": 840, |
| "step_time": 4.861015654052608 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.5, |
| "completions/max_terminated_length": 72.5, |
| "completions/mean_length": 50.70078125, |
| "completions/mean_terminated_length": 50.70078125, |
| "completions/min_length": 28.3, |
| "completions/min_terminated_length": 28.3, |
| "entropy": 0.004034929322733661, |
| "epoch": 2.1963824289405687, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 1.5625, |
| "learning_rate": 9.915100000000001e-06, |
| "loss": 0.0023, |
| "num_tokens": 46876409.0, |
| "reward": 0.34218751937150954, |
| "reward_std": 0.4044925257563591, |
| "rewards/reward_accuracy/mean": 0.2421875, |
| "rewards/reward_accuracy/std": 0.4044925257563591, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.483259344100952, |
| "sampling/importance_sampling_ratio/mean": 1.0043853878974915, |
| "sampling/importance_sampling_ratio/min": 0.6371841877698898, |
| "sampling/sampling_logp_difference/max": 0.49374919533729555, |
| "sampling/sampling_logp_difference/mean": 0.0006545575655763969, |
| "step": 850, |
| "step_time": 4.978616300155409 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 75.4, |
| "completions/max_terminated_length": 75.4, |
| "completions/mean_length": 50.634375, |
| "completions/mean_terminated_length": 50.634375, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.005173574784203083, |
| "epoch": 2.2222222222222223, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.98828125, |
| "learning_rate": 9.9141e-06, |
| "loss": 0.0013, |
| "num_tokens": 47426741.0, |
| "reward": 0.3539062708616257, |
| "reward_std": 0.4208608865737915, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4208608865737915, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4205081343650818, |
| "sampling/importance_sampling_ratio/mean": 1.001450741291046, |
| "sampling/importance_sampling_ratio/min": 0.6549598395824432, |
| "sampling/sampling_logp_difference/max": 0.5021638929843902, |
| "sampling/sampling_logp_difference/mean": 0.0008585742907598615, |
| "step": 860, |
| "step_time": 5.093159057153389 |
| }, |
| { |
| "clip_ratio/high_max": 7.077352493070065e-05, |
| "clip_ratio/high_mean": 3.538676246535033e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.538676246535033e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 49.87265625, |
| "completions/mean_terminated_length": 49.87265625, |
| "completions/min_length": 25.9, |
| "completions/min_terminated_length": 25.9, |
| "entropy": 0.004926550975142163, |
| "epoch": 2.248062015503876, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.5546875, |
| "learning_rate": 9.913100000000002e-06, |
| "loss": 0.0008, |
| "num_tokens": 47974474.0, |
| "reward": 0.3937500223517418, |
| "reward_std": 0.4398671418428421, |
| "rewards/reward_accuracy/mean": 0.29375, |
| "rewards/reward_accuracy/std": 0.4398671418428421, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.352151346206665, |
| "sampling/importance_sampling_ratio/mean": 1.0000874996185303, |
| "sampling/importance_sampling_ratio/min": 0.7131705969572067, |
| "sampling/sampling_logp_difference/max": 0.4323227047920227, |
| "sampling/sampling_logp_difference/mean": 0.0007560940168332309, |
| "step": 870, |
| "step_time": 4.945309548778459 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 75.5, |
| "completions/max_terminated_length": 75.5, |
| "completions/mean_length": 51.06015625, |
| "completions/mean_terminated_length": 51.06015625, |
| "completions/min_length": 27.1, |
| "completions/min_terminated_length": 27.1, |
| "entropy": 0.005470211217470933, |
| "epoch": 2.2739018087855296, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.912100000000001e-06, |
| "loss": 0.0013, |
| "num_tokens": 48527135.0, |
| "reward": 0.3507812723517418, |
| "reward_std": 0.42368650138378144, |
| "rewards/reward_accuracy/mean": 0.25078125, |
| "rewards/reward_accuracy/std": 0.4236864984035492, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5907588958740235, |
| "sampling/importance_sampling_ratio/mean": 1.0005271315574646, |
| "sampling/importance_sampling_ratio/min": 0.5409280881285667, |
| "sampling/sampling_logp_difference/max": 0.7270253717899322, |
| "sampling/sampling_logp_difference/mean": 0.0007456279097823426, |
| "step": 880, |
| "step_time": 5.098150842729956 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 51.8328125, |
| "completions/mean_terminated_length": 51.8328125, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.003988285773266398, |
| "epoch": 2.2997416020671837, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.59765625, |
| "learning_rate": 9.9111e-06, |
| "loss": 0.0014, |
| "num_tokens": 49082713.0, |
| "reward": 0.3710937723517418, |
| "reward_std": 0.4353772044181824, |
| "rewards/reward_accuracy/mean": 0.27109375, |
| "rewards/reward_accuracy/std": 0.4353772014379501, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.755838966369629, |
| "sampling/importance_sampling_ratio/mean": 1.0037154614925385, |
| "sampling/importance_sampling_ratio/min": 0.6124966740608215, |
| "sampling/sampling_logp_difference/max": 0.7130098938941956, |
| "sampling/sampling_logp_difference/mean": 0.0008078610466327518, |
| "step": 890, |
| "step_time": 4.972993276664056 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.5, |
| "completions/max_terminated_length": 71.5, |
| "completions/mean_length": 48.13984375, |
| "completions/mean_terminated_length": 48.13984375, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.003315521332115168, |
| "epoch": 2.3255813953488373, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.255859375, |
| "learning_rate": 9.9101e-06, |
| "loss": 0.002, |
| "num_tokens": 49624212.0, |
| "reward": 0.3164062723517418, |
| "reward_std": 0.4022886291146278, |
| "rewards/reward_accuracy/mean": 0.21640625, |
| "rewards/reward_accuracy/std": 0.4022886291146278, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5042515873908997, |
| "sampling/importance_sampling_ratio/mean": 0.9981481373310089, |
| "sampling/importance_sampling_ratio/min": 0.547298926115036, |
| "sampling/sampling_logp_difference/max": 0.6289287447929383, |
| "sampling/sampling_logp_difference/mean": 0.0007885636936407536, |
| "step": 900, |
| "step_time": 4.973210783745162 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.9, |
| "completions/max_terminated_length": 72.9, |
| "completions/mean_length": 49.68359375, |
| "completions/mean_terminated_length": 49.68359375, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.003464171539781091, |
| "epoch": 2.351421188630491, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9091e-06, |
| "loss": 0.0036, |
| "num_tokens": 50170943.0, |
| "reward": 0.4437109589576721, |
| "reward_std": 0.469940060377121, |
| "rewards/reward_accuracy/mean": 0.34375, |
| "rewards/reward_accuracy/std": 0.46991961896419526, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.5933055400848388, |
| "sampling/importance_sampling_ratio/mean": 0.9917938590049744, |
| "sampling/importance_sampling_ratio/min": 0.6466727465391159, |
| "sampling/sampling_logp_difference/max": 0.6556817173957825, |
| "sampling/sampling_logp_difference/mean": 0.0008543060801457613, |
| "step": 910, |
| "step_time": 4.965822433726862 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.4, |
| "completions/max_terminated_length": 72.4, |
| "completions/mean_length": 49.68046875, |
| "completions/mean_terminated_length": 49.68046875, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.0031384182067995424, |
| "epoch": 2.3772609819121446, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.908100000000001e-06, |
| "loss": 0.0011, |
| "num_tokens": 50719638.0, |
| "reward": 0.3328125223517418, |
| "reward_std": 0.4075076162815094, |
| "rewards/reward_accuracy/mean": 0.2328125, |
| "rewards/reward_accuracy/std": 0.40750761330127716, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2286959648132325, |
| "sampling/importance_sampling_ratio/mean": 0.9975734174251556, |
| "sampling/importance_sampling_ratio/min": 0.7111031293869019, |
| "sampling/sampling_logp_difference/max": 0.36818747520446776, |
| "sampling/sampling_logp_difference/mean": 0.00046370933996513485, |
| "step": 920, |
| "step_time": 4.932787936646491 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 101.0, |
| "completions/max_terminated_length": 101.0, |
| "completions/mean_length": 51.33828125, |
| "completions/mean_terminated_length": 51.33828125, |
| "completions/min_length": 28.3, |
| "completions/min_terminated_length": 28.3, |
| "entropy": 0.004764567659549357, |
| "epoch": 2.4031007751937983, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9071e-06, |
| "loss": -0.0025, |
| "num_tokens": 51272327.0, |
| "reward": 0.3281250223517418, |
| "reward_std": 0.41614367961883547, |
| "rewards/reward_accuracy/mean": 0.228125, |
| "rewards/reward_accuracy/std": 0.4161436766386032, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3741876244544984, |
| "sampling/importance_sampling_ratio/mean": 0.9987139880657196, |
| "sampling/importance_sampling_ratio/min": 0.4937171578407288, |
| "sampling/sampling_logp_difference/max": 0.7669991672039032, |
| "sampling/sampling_logp_difference/mean": 0.0008919914020225405, |
| "step": 930, |
| "step_time": 5.37749073964078 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 50.10078125, |
| "completions/mean_terminated_length": 50.10078125, |
| "completions/min_length": 28.1, |
| "completions/min_terminated_length": 28.1, |
| "entropy": 0.0036436536125620477, |
| "epoch": 2.4289405684754524, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.69921875, |
| "learning_rate": 9.9061e-06, |
| "loss": 0.001, |
| "num_tokens": 51821696.0, |
| "reward": 0.4250000223517418, |
| "reward_std": 0.44909388422966, |
| "rewards/reward_accuracy/mean": 0.325, |
| "rewards/reward_accuracy/std": 0.44909388422966, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4936904191970826, |
| "sampling/importance_sampling_ratio/mean": 0.9937232673168183, |
| "sampling/importance_sampling_ratio/min": 0.6535285145044327, |
| "sampling/sampling_logp_difference/max": 0.5220021665096283, |
| "sampling/sampling_logp_difference/mean": 0.000727060143253766, |
| "step": 940, |
| "step_time": 4.914706579199992 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.4, |
| "completions/max_terminated_length": 74.4, |
| "completions/mean_length": 48.12421875, |
| "completions/mean_terminated_length": 48.12421875, |
| "completions/min_length": 28.4, |
| "completions/min_terminated_length": 28.4, |
| "entropy": 0.004686664654582273, |
| "epoch": 2.454780361757106, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 1.3671875, |
| "learning_rate": 9.905100000000001e-06, |
| "loss": 0.0028, |
| "num_tokens": 52364503.0, |
| "reward": 0.39296877235174177, |
| "reward_std": 0.4373572483658791, |
| "rewards/reward_accuracy/mean": 0.29296875, |
| "rewards/reward_accuracy/std": 0.4373572483658791, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.325402283668518, |
| "sampling/importance_sampling_ratio/mean": 0.9960103809833527, |
| "sampling/importance_sampling_ratio/min": 0.6009136497974396, |
| "sampling/sampling_logp_difference/max": 0.5738130807876587, |
| "sampling/sampling_logp_difference/mean": 0.0007618420117069036, |
| "step": 950, |
| "step_time": 4.9815029253717515 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 89.5, |
| "completions/max_terminated_length": 89.5, |
| "completions/mean_length": 51.3140625, |
| "completions/mean_terminated_length": 51.3140625, |
| "completions/min_length": 27.4, |
| "completions/min_terminated_length": 27.4, |
| "entropy": 0.004903959179318918, |
| "epoch": 2.4806201550387597, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9041e-06, |
| "loss": 0.0047, |
| "num_tokens": 52917537.0, |
| "reward": 0.3242187723517418, |
| "reward_std": 0.4099811136722565, |
| "rewards/reward_accuracy/mean": 0.22421875, |
| "rewards/reward_accuracy/std": 0.4099811136722565, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4237702250480653, |
| "sampling/importance_sampling_ratio/mean": 1.0021498858928681, |
| "sampling/importance_sampling_ratio/min": 0.7194644331932067, |
| "sampling/sampling_logp_difference/max": 0.49750725626945497, |
| "sampling/sampling_logp_difference/mean": 0.0007224667177069932, |
| "step": 960, |
| "step_time": 5.243556299246848 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 50.378125, |
| "completions/mean_terminated_length": 50.378125, |
| "completions/min_length": 26.9, |
| "completions/min_terminated_length": 26.9, |
| "entropy": 0.0028216493547006394, |
| "epoch": 2.5064599483204133, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 0.3671875, |
| "learning_rate": 9.903100000000002e-06, |
| "loss": -0.0014, |
| "num_tokens": 53468013.0, |
| "reward": 0.3609375223517418, |
| "reward_std": 0.42067247778177264, |
| "rewards/reward_accuracy/mean": 0.2609375, |
| "rewards/reward_accuracy/std": 0.42067247778177264, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4731903076171875, |
| "sampling/importance_sampling_ratio/mean": 1.0004703640937804, |
| "sampling/importance_sampling_ratio/min": 0.7312530905008316, |
| "sampling/sampling_logp_difference/max": 0.44934140592813493, |
| "sampling/sampling_logp_difference/mean": 0.00047647648316342386, |
| "step": 970, |
| "step_time": 4.8967843315331265 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 49.4984375, |
| "completions/mean_terminated_length": 49.4984375, |
| "completions/min_length": 26.3, |
| "completions/min_terminated_length": 26.3, |
| "entropy": 0.0025121401029537084, |
| "epoch": 2.532299741602067, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.902100000000001e-06, |
| "loss": 0.0031, |
| "num_tokens": 54014011.0, |
| "reward": 0.3937500238418579, |
| "reward_std": 0.44737596809864044, |
| "rewards/reward_accuracy/mean": 0.29375, |
| "rewards/reward_accuracy/std": 0.44737596809864044, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.364039421081543, |
| "sampling/importance_sampling_ratio/mean": 0.9970609724521637, |
| "sampling/importance_sampling_ratio/min": 0.7420731633901596, |
| "sampling/sampling_logp_difference/max": 0.5713446628302336, |
| "sampling/sampling_logp_difference/mean": 0.00045454270602931504, |
| "step": 980, |
| "step_time": 5.019445966300554 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 82.0, |
| "completions/max_terminated_length": 82.0, |
| "completions/mean_length": 49.6828125, |
| "completions/mean_terminated_length": 49.6828125, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.004109046176199627, |
| "epoch": 2.558139534883721, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9011e-06, |
| "loss": -0.0002, |
| "num_tokens": 54560445.0, |
| "reward": 0.3585937708616257, |
| "reward_std": 0.42351907938718797, |
| "rewards/reward_accuracy/mean": 0.25859375, |
| "rewards/reward_accuracy/std": 0.42351907938718797, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.382976496219635, |
| "sampling/importance_sampling_ratio/mean": 0.9986992955207825, |
| "sampling/importance_sampling_ratio/min": 0.7391488879919053, |
| "sampling/sampling_logp_difference/max": 0.4023936866782606, |
| "sampling/sampling_logp_difference/mean": 0.0005228334377534339, |
| "step": 990, |
| "step_time": 5.142811875510961 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.8, |
| "completions/max_terminated_length": 73.8, |
| "completions/mean_length": 50.01171875, |
| "completions/mean_terminated_length": 50.01171875, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.0028876276594019144, |
| "epoch": 2.5839793281653747, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.55859375, |
| "learning_rate": 9.9001e-06, |
| "loss": -0.0028, |
| "num_tokens": 55109476.0, |
| "reward": 0.3601562723517418, |
| "reward_std": 0.42405185103416443, |
| "rewards/reward_accuracy/mean": 0.26015625, |
| "rewards/reward_accuracy/std": 0.42405185103416443, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.331214225292206, |
| "sampling/importance_sampling_ratio/mean": 0.9985450327396392, |
| "sampling/importance_sampling_ratio/min": 0.7490552484989166, |
| "sampling/sampling_logp_difference/max": 0.3235751509666443, |
| "sampling/sampling_logp_difference/mean": 0.00036550726945279167, |
| "step": 1000, |
| "step_time": 4.906477443128824 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.4, |
| "completions/max_terminated_length": 73.4, |
| "completions/mean_length": 50.5890625, |
| "completions/mean_terminated_length": 50.5890625, |
| "completions/min_length": 29.0, |
| "completions/min_terminated_length": 29.0, |
| "entropy": 0.0031119672863496816, |
| "epoch": 2.6098191214470283, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 3.515625, |
| "learning_rate": 9.8991e-06, |
| "loss": 0.003, |
| "num_tokens": 55660702.0, |
| "reward": 0.4015625223517418, |
| "reward_std": 0.4492542505264282, |
| "rewards/reward_accuracy/mean": 0.3015625, |
| "rewards/reward_accuracy/std": 0.4492542505264282, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3860404372215271, |
| "sampling/importance_sampling_ratio/mean": 0.9982279121875763, |
| "sampling/importance_sampling_ratio/min": 0.7215697735548019, |
| "sampling/sampling_logp_difference/max": 0.4866408586502075, |
| "sampling/sampling_logp_difference/mean": 0.0005690376630809624, |
| "step": 1010, |
| "step_time": 5.045411675237119 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 51.665625, |
| "completions/mean_terminated_length": 51.665625, |
| "completions/min_length": 23.2, |
| "completions/min_terminated_length": 23.2, |
| "entropy": 0.0032653684182150757, |
| "epoch": 2.6356589147286824, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 3.984375, |
| "learning_rate": 9.898100000000001e-06, |
| "loss": -0.0089, |
| "num_tokens": 56216498.0, |
| "reward": 0.3351562708616257, |
| "reward_std": 0.40758039206266405, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.40758039206266405, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5962645649909972, |
| "sampling/importance_sampling_ratio/mean": 1.0002167105674744, |
| "sampling/importance_sampling_ratio/min": 0.6659949719905853, |
| "sampling/sampling_logp_difference/max": 0.5241124391555786, |
| "sampling/sampling_logp_difference/mean": 0.0007451553829014301, |
| "step": 1020, |
| "step_time": 4.956005468964577 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 51.215625, |
| "completions/mean_terminated_length": 51.215625, |
| "completions/min_length": 28.0, |
| "completions/min_terminated_length": 28.0, |
| "entropy": 0.0033274123690716804, |
| "epoch": 2.661498708010336, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8971e-06, |
| "loss": 0.0002, |
| "num_tokens": 56768518.0, |
| "reward": 0.36875001937150953, |
| "reward_std": 0.4187498867511749, |
| "rewards/reward_accuracy/mean": 0.26875, |
| "rewards/reward_accuracy/std": 0.4187498867511749, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3949612617492675, |
| "sampling/importance_sampling_ratio/mean": 0.9984726011753082, |
| "sampling/importance_sampling_ratio/min": 0.5531232237815857, |
| "sampling/sampling_logp_difference/max": 0.6857089042663574, |
| "sampling/sampling_logp_difference/mean": 0.000631844880990684, |
| "step": 1030, |
| "step_time": 4.996687895478681 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.7, |
| "completions/max_terminated_length": 74.7, |
| "completions/mean_length": 48.21328125, |
| "completions/mean_terminated_length": 48.21328125, |
| "completions/min_length": 26.3, |
| "completions/min_terminated_length": 26.3, |
| "entropy": 0.004576992360671284, |
| "epoch": 2.6873385012919897, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8961e-06, |
| "loss": -0.0022, |
| "num_tokens": 57309407.0, |
| "reward": 0.4070312723517418, |
| "reward_std": 0.4461204826831818, |
| "rewards/reward_accuracy/mean": 0.30703125, |
| "rewards/reward_accuracy/std": 0.44612047970294955, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3473424553871154, |
| "sampling/importance_sampling_ratio/mean": 0.9989716827869415, |
| "sampling/importance_sampling_ratio/min": 0.6753489583730697, |
| "sampling/sampling_logp_difference/max": 0.40977575778961184, |
| "sampling/sampling_logp_difference/mean": 0.0006006711351801642, |
| "step": 1040, |
| "step_time": 5.012189233559184 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.8, |
| "completions/max_terminated_length": 71.8, |
| "completions/mean_length": 50.65859375, |
| "completions/mean_terminated_length": 50.65859375, |
| "completions/min_length": 27.8, |
| "completions/min_terminated_length": 27.8, |
| "entropy": 0.005480249292668304, |
| "epoch": 2.7131782945736433, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 1.3046875, |
| "learning_rate": 9.895100000000001e-06, |
| "loss": 0.0055, |
| "num_tokens": 57861066.0, |
| "reward": 0.35386721044778824, |
| "reward_std": 0.42074478417634964, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4207176402211189, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.7528934597969055, |
| "sampling/importance_sampling_ratio/mean": 1.0031718134880065, |
| "sampling/importance_sampling_ratio/min": 0.6231196284294128, |
| "sampling/sampling_logp_difference/max": 0.5383359670639039, |
| "sampling/sampling_logp_difference/mean": 0.0008972429670393467, |
| "step": 1050, |
| "step_time": 4.9472716975724325 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.4, |
| "completions/max_terminated_length": 72.4, |
| "completions/mean_length": 48.271875, |
| "completions/mean_terminated_length": 48.271875, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.005814611278037773, |
| "epoch": 2.739018087855297, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 3.015625, |
| "learning_rate": 9.8941e-06, |
| "loss": -0.005, |
| "num_tokens": 58402278.0, |
| "reward": 0.3898437738418579, |
| "reward_std": 0.44406315982341765, |
| "rewards/reward_accuracy/mean": 0.28984375, |
| "rewards/reward_accuracy/std": 0.44406315982341765, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5818143248558045, |
| "sampling/importance_sampling_ratio/mean": 1.0022391736507417, |
| "sampling/importance_sampling_ratio/min": 0.6759881168603897, |
| "sampling/sampling_logp_difference/max": 0.43719334006309507, |
| "sampling/sampling_logp_difference/mean": 0.0007514142023865133, |
| "step": 1060, |
| "step_time": 4.970276336208917 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.8, |
| "completions/max_terminated_length": 72.8, |
| "completions/mean_length": 51.7171875, |
| "completions/mean_terminated_length": 51.7171875, |
| "completions/min_length": 28.4, |
| "completions/min_terminated_length": 28.4, |
| "entropy": 0.006372165818902431, |
| "epoch": 2.764857881136951, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.83203125, |
| "learning_rate": 9.8931e-06, |
| "loss": -0.0005, |
| "num_tokens": 58957252.0, |
| "reward": 0.28828126937150955, |
| "reward_std": 0.3807651102542877, |
| "rewards/reward_accuracy/mean": 0.18828125, |
| "rewards/reward_accuracy/std": 0.3807651102542877, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.555271315574646, |
| "sampling/importance_sampling_ratio/mean": 1.001236653327942, |
| "sampling/importance_sampling_ratio/min": 0.6131647229194641, |
| "sampling/sampling_logp_difference/max": 0.5463376343250275, |
| "sampling/sampling_logp_difference/mean": 0.0008582875350839458, |
| "step": 1070, |
| "step_time": 4.908182834181934 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 69.6, |
| "completions/max_terminated_length": 69.6, |
| "completions/mean_length": 50.39765625, |
| "completions/mean_terminated_length": 50.39765625, |
| "completions/min_length": 27.4, |
| "completions/min_terminated_length": 27.4, |
| "entropy": 0.0053042626968817785, |
| "epoch": 2.7906976744186047, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.892100000000001e-06, |
| "loss": 0.0012, |
| "num_tokens": 59508089.0, |
| "reward": 0.40000001937150953, |
| "reward_std": 0.41482011377811434, |
| "rewards/reward_accuracy/mean": 0.3, |
| "rewards/reward_accuracy/std": 0.41482011377811434, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.364490520954132, |
| "sampling/importance_sampling_ratio/mean": 0.9992563486099243, |
| "sampling/importance_sampling_ratio/min": 0.597221502661705, |
| "sampling/sampling_logp_difference/max": 0.585259860754013, |
| "sampling/sampling_logp_difference/mean": 0.0008712529030162841, |
| "step": 1080, |
| "step_time": 4.8691873519448565 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.4, |
| "completions/max_terminated_length": 73.4, |
| "completions/mean_length": 52.11875, |
| "completions/mean_terminated_length": 52.11875, |
| "completions/min_length": 29.1, |
| "completions/min_terminated_length": 29.1, |
| "entropy": 0.00557758176146308, |
| "epoch": 2.8165374677002584, |
| "frac_reward_zero_std": 0.91875, |
| "grad_norm": 2.109375, |
| "learning_rate": 9.891100000000001e-06, |
| "loss": -0.004, |
| "num_tokens": 60062313.0, |
| "reward": 0.3484375223517418, |
| "reward_std": 0.41986130028963087, |
| "rewards/reward_accuracy/mean": 0.2484375, |
| "rewards/reward_accuracy/std": 0.41986129730939864, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5915640234947204, |
| "sampling/importance_sampling_ratio/mean": 0.9986651837825775, |
| "sampling/importance_sampling_ratio/min": 0.5135205041617155, |
| "sampling/sampling_logp_difference/max": 0.8623277723789216, |
| "sampling/sampling_logp_difference/mean": 0.0012016980559565126, |
| "step": 1090, |
| "step_time": 5.0061027761083094 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.1, |
| "completions/max_terminated_length": 74.1, |
| "completions/mean_length": 52.57109375, |
| "completions/mean_terminated_length": 52.57109375, |
| "completions/min_length": 24.7, |
| "completions/min_terminated_length": 24.7, |
| "entropy": 0.004560436031533754, |
| "epoch": 2.842377260981912, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8901e-06, |
| "loss": -0.0005, |
| "num_tokens": 60619468.0, |
| "reward": 0.40078127235174177, |
| "reward_std": 0.44620595276355746, |
| "rewards/reward_accuracy/mean": 0.30078125, |
| "rewards/reward_accuracy/std": 0.4462059497833252, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5040345311164856, |
| "sampling/importance_sampling_ratio/mean": 0.9936056137084961, |
| "sampling/importance_sampling_ratio/min": 0.55034848600626, |
| "sampling/sampling_logp_difference/max": 0.6696406245231629, |
| "sampling/sampling_logp_difference/mean": 0.0009015202464070171, |
| "step": 1100, |
| "step_time": 4.981047308491543 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.6, |
| "completions/max_terminated_length": 72.6, |
| "completions/mean_length": 51.5953125, |
| "completions/mean_terminated_length": 51.5953125, |
| "completions/min_length": 27.6, |
| "completions/min_terminated_length": 27.6, |
| "entropy": 0.005168237066391157, |
| "epoch": 2.8682170542635657, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8891e-06, |
| "loss": -0.003, |
| "num_tokens": 61174550.0, |
| "reward": 0.3390625223517418, |
| "reward_std": 0.42035396993160246, |
| "rewards/reward_accuracy/mean": 0.2390625, |
| "rewards/reward_accuracy/std": 0.42035396993160246, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.509937834739685, |
| "sampling/importance_sampling_ratio/mean": 1.0003831326961516, |
| "sampling/importance_sampling_ratio/min": 0.6416405498981476, |
| "sampling/sampling_logp_difference/max": 0.5356657385826111, |
| "sampling/sampling_logp_difference/mean": 0.000739611077005975, |
| "step": 1110, |
| "step_time": 5.032000749721192 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.8, |
| "completions/max_terminated_length": 74.8, |
| "completions/mean_length": 51.48515625, |
| "completions/mean_terminated_length": 51.48515625, |
| "completions/min_length": 27.8, |
| "completions/min_terminated_length": 27.8, |
| "entropy": 0.0035099805834761357, |
| "epoch": 2.8940568475452197, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.888100000000001e-06, |
| "loss": -0.0017, |
| "num_tokens": 61730123.0, |
| "reward": 0.43750002384185793, |
| "reward_std": 0.4674528777599335, |
| "rewards/reward_accuracy/mean": 0.3375, |
| "rewards/reward_accuracy/std": 0.4674528777599335, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4031189680099487, |
| "sampling/importance_sampling_ratio/mean": 0.9964274883270263, |
| "sampling/importance_sampling_ratio/min": 0.6828254580497741, |
| "sampling/sampling_logp_difference/max": 0.5191488467156887, |
| "sampling/sampling_logp_difference/mean": 0.0007021360335784266, |
| "step": 1120, |
| "step_time": 5.052406533597969 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.9, |
| "completions/max_terminated_length": 74.9, |
| "completions/mean_length": 48.8390625, |
| "completions/mean_terminated_length": 48.8390625, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.004019141541721183, |
| "epoch": 2.9198966408268734, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8871e-06, |
| "loss": -0.0055, |
| "num_tokens": 62273837.0, |
| "reward": 0.3601562723517418, |
| "reward_std": 0.4270425260066986, |
| "rewards/reward_accuracy/mean": 0.26015625, |
| "rewards/reward_accuracy/std": 0.4270425230264664, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5247862100601197, |
| "sampling/importance_sampling_ratio/mean": 1.0024777114391328, |
| "sampling/importance_sampling_ratio/min": 0.622413232922554, |
| "sampling/sampling_logp_difference/max": 0.6054848313331604, |
| "sampling/sampling_logp_difference/mean": 0.0007624067380675115, |
| "step": 1130, |
| "step_time": 4.958793773851357 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.4, |
| "completions/max_terminated_length": 74.4, |
| "completions/mean_length": 50.09453125, |
| "completions/mean_terminated_length": 50.09453125, |
| "completions/min_length": 25.1, |
| "completions/min_terminated_length": 25.1, |
| "entropy": 0.003228264501558442, |
| "epoch": 2.945736434108527, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 2.46875, |
| "learning_rate": 9.8861e-06, |
| "loss": -0.0043, |
| "num_tokens": 62821022.0, |
| "reward": 0.3632812723517418, |
| "reward_std": 0.4250364124774933, |
| "rewards/reward_accuracy/mean": 0.26328125, |
| "rewards/reward_accuracy/std": 0.4250364124774933, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7817544341087341, |
| "sampling/importance_sampling_ratio/mean": 1.0011677145957947, |
| "sampling/importance_sampling_ratio/min": 0.5583647519350052, |
| "sampling/sampling_logp_difference/max": 0.8132681012153625, |
| "sampling/sampling_logp_difference/mean": 0.0008545911579858512, |
| "step": 1140, |
| "step_time": 5.003933532489464 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 105.1, |
| "completions/max_terminated_length": 105.1, |
| "completions/mean_length": 50.51640625, |
| "completions/mean_terminated_length": 50.51640625, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.0042277246706362345, |
| "epoch": 2.971576227390181, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.885100000000001e-06, |
| "loss": -0.0017, |
| "num_tokens": 63370035.0, |
| "reward": 0.3875000238418579, |
| "reward_std": 0.44623408317565916, |
| "rewards/reward_accuracy/mean": 0.2875, |
| "rewards/reward_accuracy/std": 0.44623408019542693, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6375315070152283, |
| "sampling/importance_sampling_ratio/mean": 0.9992500066757202, |
| "sampling/importance_sampling_ratio/min": 0.6481870532035827, |
| "sampling/sampling_logp_difference/max": 0.5402241706848144, |
| "sampling/sampling_logp_difference/mean": 0.0008037368796067312, |
| "step": 1150, |
| "step_time": 5.420169553207233 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.8, |
| "completions/max_terminated_length": 72.8, |
| "completions/mean_length": 49.8125, |
| "completions/mean_terminated_length": 49.8125, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.004112232074294298, |
| "epoch": 2.9974160206718348, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8841e-06, |
| "loss": 0.0044, |
| "num_tokens": 63917907.0, |
| "reward": 0.3546875201165676, |
| "reward_std": 0.39511207342147825, |
| "rewards/reward_accuracy/mean": 0.2546875, |
| "rewards/reward_accuracy/std": 0.39511207342147825, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5641756892204284, |
| "sampling/importance_sampling_ratio/mean": 0.9895751059055329, |
| "sampling/importance_sampling_ratio/min": 0.5473731517791748, |
| "sampling/sampling_logp_difference/max": 0.7250224351882935, |
| "sampling/sampling_logp_difference/mean": 0.001128919783513993, |
| "step": 1160, |
| "step_time": 4.9451073508476835 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.3950893480796367e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.3950893480796367e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 47.575, |
| "completions/mean_terminated_length": 47.575, |
| "completions/min_length": 25.1, |
| "completions/min_terminated_length": 25.1, |
| "entropy": 0.004587455904038506, |
| "epoch": 3.0232558139534884, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 1.5078125, |
| "learning_rate": 9.8831e-06, |
| "loss": -0.0021, |
| "num_tokens": 64455747.0, |
| "reward": 0.4210937738418579, |
| "reward_std": 0.4473822325468063, |
| "rewards/reward_accuracy/mean": 0.32109375, |
| "rewards/reward_accuracy/std": 0.4473822295665741, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.362823486328125, |
| "sampling/importance_sampling_ratio/mean": 0.9969162106513977, |
| "sampling/importance_sampling_ratio/min": 0.628165665268898, |
| "sampling/sampling_logp_difference/max": 0.5191500544548034, |
| "sampling/sampling_logp_difference/mean": 0.0007282782244146802, |
| "step": 1170, |
| "step_time": 4.879565683775581 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 76.3, |
| "completions/max_terminated_length": 76.3, |
| "completions/mean_length": 52.0484375, |
| "completions/mean_terminated_length": 52.0484375, |
| "completions/min_length": 28.8, |
| "completions/min_terminated_length": 28.8, |
| "entropy": 0.005462613153940765, |
| "epoch": 3.049095607235142, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.412109375, |
| "learning_rate": 9.882100000000001e-06, |
| "loss": -0.0015, |
| "num_tokens": 65012737.0, |
| "reward": 0.34921876937150953, |
| "reward_std": 0.4051841706037521, |
| "rewards/reward_accuracy/mean": 0.24921875, |
| "rewards/reward_accuracy/std": 0.4051841706037521, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7264459609985352, |
| "sampling/importance_sampling_ratio/mean": 1.000433337688446, |
| "sampling/importance_sampling_ratio/min": 0.5039217889308929, |
| "sampling/sampling_logp_difference/max": 0.7098558783531189, |
| "sampling/sampling_logp_difference/mean": 0.0009893351787468418, |
| "step": 1180, |
| "step_time": 5.012715068086981 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.8, |
| "completions/max_terminated_length": 73.8, |
| "completions/mean_length": 50.7203125, |
| "completions/mean_terminated_length": 50.7203125, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.005655257676698966, |
| "epoch": 3.0749354005167957, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 2.203125, |
| "learning_rate": 9.881100000000001e-06, |
| "loss": -0.0068, |
| "num_tokens": 65563635.0, |
| "reward": 0.35156252086162565, |
| "reward_std": 0.42687056958675385, |
| "rewards/reward_accuracy/mean": 0.2515625, |
| "rewards/reward_accuracy/std": 0.4268705666065216, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7113429188728333, |
| "sampling/importance_sampling_ratio/mean": 1.001689612865448, |
| "sampling/importance_sampling_ratio/min": 0.525804440677166, |
| "sampling/sampling_logp_difference/max": 0.7514476180076599, |
| "sampling/sampling_logp_difference/mean": 0.0012438131190720015, |
| "step": 1190, |
| "step_time": 5.029049736214802 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.4, |
| "completions/max_terminated_length": 72.4, |
| "completions/mean_length": 50.5578125, |
| "completions/mean_terminated_length": 50.5578125, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.004373838145693298, |
| "epoch": 3.10077519379845, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.880100000000002e-06, |
| "loss": 0.0003, |
| "num_tokens": 66113085.0, |
| "reward": 0.3608984604477882, |
| "reward_std": 0.4347838968038559, |
| "rewards/reward_accuracy/mean": 0.2609375, |
| "rewards/reward_accuracy/std": 0.43475536704063417, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.4147764205932618, |
| "sampling/importance_sampling_ratio/mean": 0.9951588690280915, |
| "sampling/importance_sampling_ratio/min": 0.6491352319717407, |
| "sampling/sampling_logp_difference/max": 0.5916034750640392, |
| "sampling/sampling_logp_difference/mean": 0.0007282761769602075, |
| "step": 1200, |
| "step_time": 4.957652619294822 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.3, |
| "completions/max_terminated_length": 70.3, |
| "completions/mean_length": 50.04296875, |
| "completions/mean_terminated_length": 50.04296875, |
| "completions/min_length": 25.9, |
| "completions/min_terminated_length": 25.9, |
| "entropy": 0.005084942981557106, |
| "epoch": 3.1266149870801034, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8791e-06, |
| "loss": -0.0037, |
| "num_tokens": 66660756.0, |
| "reward": 0.3749609589576721, |
| "reward_std": 0.41621437221765517, |
| "rewards/reward_accuracy/mean": 0.275, |
| "rewards/reward_accuracy/std": 0.41618199795484545, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.5692046284675598, |
| "sampling/importance_sampling_ratio/mean": 0.9927727580070496, |
| "sampling/importance_sampling_ratio/min": 0.5717119798064232, |
| "sampling/sampling_logp_difference/max": 0.7682465333491564, |
| "sampling/sampling_logp_difference/mean": 0.0009768518484634114, |
| "step": 1210, |
| "step_time": 4.9040765423560515 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.5, |
| "completions/max_terminated_length": 70.5, |
| "completions/mean_length": 50.1421875, |
| "completions/mean_terminated_length": 50.1421875, |
| "completions/min_length": 28.0, |
| "completions/min_terminated_length": 28.0, |
| "entropy": 0.005041458871710347, |
| "epoch": 3.152454780361757, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.878100000000001e-06, |
| "loss": -0.0054, |
| "num_tokens": 67210634.0, |
| "reward": 0.34218751788139345, |
| "reward_std": 0.39799479991197584, |
| "rewards/reward_accuracy/mean": 0.2421875, |
| "rewards/reward_accuracy/std": 0.39799479991197584, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.799338686466217, |
| "sampling/importance_sampling_ratio/mean": 0.9988008856773376, |
| "sampling/importance_sampling_ratio/min": 0.5620145365595818, |
| "sampling/sampling_logp_difference/max": 0.800980019569397, |
| "sampling/sampling_logp_difference/mean": 0.0013800672313664109, |
| "step": 1220, |
| "step_time": 4.981580855417997 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.2, |
| "completions/max_terminated_length": 74.2, |
| "completions/mean_length": 48.81796875, |
| "completions/mean_terminated_length": 48.81796875, |
| "completions/min_length": 26.2, |
| "completions/min_terminated_length": 26.2, |
| "entropy": 0.003625820835986815, |
| "epoch": 3.1782945736434107, |
| "frac_reward_zero_std": 0.99375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8771e-06, |
| "loss": 0.0006, |
| "num_tokens": 67753697.0, |
| "reward": 0.3640625223517418, |
| "reward_std": 0.43385404646396636, |
| "rewards/reward_accuracy/mean": 0.2640625, |
| "rewards/reward_accuracy/std": 0.43385404646396636, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.360518229007721, |
| "sampling/importance_sampling_ratio/mean": 0.9997088789939881, |
| "sampling/importance_sampling_ratio/min": 0.6651515632867813, |
| "sampling/sampling_logp_difference/max": 0.47682552337646483, |
| "sampling/sampling_logp_difference/mean": 0.0005707579126465135, |
| "step": 1230, |
| "step_time": 4.932098086224869 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.5, |
| "completions/max_terminated_length": 71.5, |
| "completions/mean_length": 51.275, |
| "completions/mean_terminated_length": 51.275, |
| "completions/min_length": 27.1, |
| "completions/min_terminated_length": 27.1, |
| "entropy": 0.004367303054823424, |
| "epoch": 3.2041343669250644, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8761e-06, |
| "loss": -0.0038, |
| "num_tokens": 68306105.0, |
| "reward": 0.34531252086162567, |
| "reward_std": 0.4209453284740448, |
| "rewards/reward_accuracy/mean": 0.2453125, |
| "rewards/reward_accuracy/std": 0.4209453254938126, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3720709204673767, |
| "sampling/importance_sampling_ratio/mean": 0.9927451491355896, |
| "sampling/importance_sampling_ratio/min": 0.6832411706447601, |
| "sampling/sampling_logp_difference/max": 0.4493452847003937, |
| "sampling/sampling_logp_difference/mean": 0.0006191601984028239, |
| "step": 1240, |
| "step_time": 4.949813662515953 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.5, |
| "completions/max_terminated_length": 72.5, |
| "completions/mean_length": 51.7375, |
| "completions/mean_terminated_length": 51.7375, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.005468618430859351, |
| "epoch": 3.2299741602067185, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.875100000000001e-06, |
| "loss": -0.0027, |
| "num_tokens": 68862569.0, |
| "reward": 0.3796875223517418, |
| "reward_std": 0.43281871974468233, |
| "rewards/reward_accuracy/mean": 0.2796875, |
| "rewards/reward_accuracy/std": 0.43281871676445005, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8078739404678346, |
| "sampling/importance_sampling_ratio/mean": 1.0022813260555268, |
| "sampling/importance_sampling_ratio/min": 0.556065059453249, |
| "sampling/sampling_logp_difference/max": 0.9088182210922241, |
| "sampling/sampling_logp_difference/mean": 0.0011126799916382879, |
| "step": 1250, |
| "step_time": 4.982934018224478 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 99.5, |
| "completions/max_terminated_length": 99.5, |
| "completions/mean_length": 50.83828125, |
| "completions/mean_terminated_length": 50.83828125, |
| "completions/min_length": 27.7, |
| "completions/min_terminated_length": 27.7, |
| "entropy": 0.006037539445605944, |
| "epoch": 3.255813953488372, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.5703125, |
| "learning_rate": 9.874100000000001e-06, |
| "loss": -0.0057, |
| "num_tokens": 69413986.0, |
| "reward": 0.38277346193790435, |
| "reward_std": 0.43870504200458527, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.4386716604232788, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.4824159860610961, |
| "sampling/importance_sampling_ratio/mean": 1.0039050102233886, |
| "sampling/importance_sampling_ratio/min": 0.631998797506094, |
| "sampling/sampling_logp_difference/max": 0.7160292953252793, |
| "sampling/sampling_logp_difference/mean": 0.0009615471528377384, |
| "step": 1260, |
| "step_time": 5.297886113054119 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 88.5, |
| "completions/max_terminated_length": 88.5, |
| "completions/mean_length": 51.365625, |
| "completions/mean_terminated_length": 51.365625, |
| "completions/min_length": 26.8, |
| "completions/min_terminated_length": 26.8, |
| "entropy": 0.006072134138958063, |
| "epoch": 3.2816537467700257, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8731e-06, |
| "loss": -0.0078, |
| "num_tokens": 69967766.0, |
| "reward": 0.3750000223517418, |
| "reward_std": 0.4338325262069702, |
| "rewards/reward_accuracy/mean": 0.275, |
| "rewards/reward_accuracy/std": 0.4338325262069702, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5845147252082825, |
| "sampling/importance_sampling_ratio/mean": 0.9967635929584503, |
| "sampling/importance_sampling_ratio/min": 0.485968653857708, |
| "sampling/sampling_logp_difference/max": 0.6965357542037964, |
| "sampling/sampling_logp_difference/mean": 0.0008902846078854054, |
| "step": 1270, |
| "step_time": 5.210583791183308 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.8, |
| "completions/max_terminated_length": 73.8, |
| "completions/mean_length": 50.834375, |
| "completions/mean_terminated_length": 50.834375, |
| "completions/min_length": 25.7, |
| "completions/min_terminated_length": 25.7, |
| "entropy": 0.0049254325131187215, |
| "epoch": 3.3074935400516794, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 2.640625, |
| "learning_rate": 9.872100000000002e-06, |
| "loss": 0.0018, |
| "num_tokens": 70518546.0, |
| "reward": 0.3828125223517418, |
| "reward_std": 0.43159347772598267, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.43159347772598267, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4964378714561462, |
| "sampling/importance_sampling_ratio/mean": 0.9995093584060669, |
| "sampling/importance_sampling_ratio/min": 0.6814517736434936, |
| "sampling/sampling_logp_difference/max": 0.5182394802570343, |
| "sampling/sampling_logp_difference/mean": 0.0007302535770577379, |
| "step": 1280, |
| "step_time": 5.0465632430277765 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.5, |
| "completions/max_terminated_length": 72.5, |
| "completions/mean_length": 50.63125, |
| "completions/mean_terminated_length": 50.63125, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.004309230094077065, |
| "epoch": 3.3333333333333335, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.2236328125, |
| "learning_rate": 9.871100000000001e-06, |
| "loss": -0.002, |
| "num_tokens": 71070162.0, |
| "reward": 0.4140625223517418, |
| "reward_std": 0.44230268001556394, |
| "rewards/reward_accuracy/mean": 0.3140625, |
| "rewards/reward_accuracy/std": 0.44230268001556394, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5678501725196838, |
| "sampling/importance_sampling_ratio/mean": 1.00375839471817, |
| "sampling/importance_sampling_ratio/min": 0.6108593970537186, |
| "sampling/sampling_logp_difference/max": 0.6590812295675278, |
| "sampling/sampling_logp_difference/mean": 0.0008943251945311203, |
| "step": 1290, |
| "step_time": 4.908462832961232 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 50.76640625, |
| "completions/mean_terminated_length": 50.76640625, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.004706332616478903, |
| "epoch": 3.359173126614987, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8701e-06, |
| "loss": 0.0019, |
| "num_tokens": 71619503.0, |
| "reward": 0.41171877086162567, |
| "reward_std": 0.4304858326911926, |
| "rewards/reward_accuracy/mean": 0.31171875, |
| "rewards/reward_accuracy/std": 0.4304858326911926, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7388676524162292, |
| "sampling/importance_sampling_ratio/mean": 1.0003097116947175, |
| "sampling/importance_sampling_ratio/min": 0.5938270330429077, |
| "sampling/sampling_logp_difference/max": 0.6479875653982162, |
| "sampling/sampling_logp_difference/mean": 0.0009545041772071272, |
| "step": 1300, |
| "step_time": 4.9357328424463045 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.5, |
| "completions/max_terminated_length": 72.5, |
| "completions/mean_length": 50.09609375, |
| "completions/mean_terminated_length": 50.09609375, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.003413056106546719, |
| "epoch": 3.3850129198966408, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 2.34375, |
| "learning_rate": 9.8691e-06, |
| "loss": -0.0009, |
| "num_tokens": 72168618.0, |
| "reward": 0.3562500223517418, |
| "reward_std": 0.41972407400608064, |
| "rewards/reward_accuracy/mean": 0.25625, |
| "rewards/reward_accuracy/std": 0.41972407400608064, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5304940581321715, |
| "sampling/importance_sampling_ratio/mean": 1.0002013325691224, |
| "sampling/importance_sampling_ratio/min": 0.7741191238164902, |
| "sampling/sampling_logp_difference/max": 0.49592754198238254, |
| "sampling/sampling_logp_difference/mean": 0.0005576418046985054, |
| "step": 1310, |
| "step_time": 4.946977186668664 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.5, |
| "completions/max_terminated_length": 71.5, |
| "completions/mean_length": 48.75703125, |
| "completions/mean_terminated_length": 48.75703125, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.0032465668846271, |
| "epoch": 3.4108527131782944, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.875, |
| "learning_rate": 9.868100000000001e-06, |
| "loss": -0.0041, |
| "num_tokens": 72712763.0, |
| "reward": 0.3679687723517418, |
| "reward_std": 0.4296472519636154, |
| "rewards/reward_accuracy/mean": 0.26796875, |
| "rewards/reward_accuracy/std": 0.4296472489833832, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3497143983840942, |
| "sampling/importance_sampling_ratio/mean": 1.0013796210289, |
| "sampling/importance_sampling_ratio/min": 0.7291356027126312, |
| "sampling/sampling_logp_difference/max": 0.4491267204284668, |
| "sampling/sampling_logp_difference/mean": 0.0004995654075173661, |
| "step": 1320, |
| "step_time": 4.971949199633673 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.6, |
| "completions/max_terminated_length": 72.6, |
| "completions/mean_length": 49.9421875, |
| "completions/mean_terminated_length": 49.9421875, |
| "completions/min_length": 25.1, |
| "completions/min_terminated_length": 25.1, |
| "entropy": 0.004414623386037419, |
| "epoch": 3.4366925064599485, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8671e-06, |
| "loss": 0.0001, |
| "num_tokens": 73261017.0, |
| "reward": 0.4375000223517418, |
| "reward_std": 0.4477925509214401, |
| "rewards/reward_accuracy/mean": 0.3375, |
| "rewards/reward_accuracy/std": 0.4477925509214401, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.1867775797843934, |
| "sampling/importance_sampling_ratio/mean": 0.9941175162792206, |
| "sampling/importance_sampling_ratio/min": 0.6091001540422439, |
| "sampling/sampling_logp_difference/max": 0.4789739564061165, |
| "sampling/sampling_logp_difference/mean": 0.0005028345447499305, |
| "step": 1330, |
| "step_time": 4.947344719059766 |
| }, |
| { |
| "clip_ratio/high_max": 2.9620854184031488e-05, |
| "clip_ratio/high_mean": 1.4810427092015744e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.4810427092015744e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.8, |
| "completions/max_terminated_length": 71.8, |
| "completions/mean_length": 52.18046875, |
| "completions/mean_terminated_length": 52.18046875, |
| "completions/min_length": 30.3, |
| "completions/min_terminated_length": 30.3, |
| "entropy": 0.0032519426931685302, |
| "epoch": 3.462532299741602, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8661e-06, |
| "loss": -0.0012, |
| "num_tokens": 73817848.0, |
| "reward": 0.3898437738418579, |
| "reward_std": 0.44870527386665343, |
| "rewards/reward_accuracy/mean": 0.28984375, |
| "rewards/reward_accuracy/std": 0.44870527386665343, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3520692706108093, |
| "sampling/importance_sampling_ratio/mean": 0.9934787333011628, |
| "sampling/importance_sampling_ratio/min": 0.6181084141135216, |
| "sampling/sampling_logp_difference/max": 0.5630139648914337, |
| "sampling/sampling_logp_difference/mean": 0.0005605273581750225, |
| "step": 1340, |
| "step_time": 4.958545914199203 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.1, |
| "completions/max_terminated_length": 72.1, |
| "completions/mean_length": 50.8296875, |
| "completions/mean_terminated_length": 50.8296875, |
| "completions/min_length": 26.6, |
| "completions/min_terminated_length": 26.6, |
| "entropy": 0.003960463025578065, |
| "epoch": 3.488372093023256, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.91015625, |
| "learning_rate": 9.865100000000001e-06, |
| "loss": -0.005, |
| "num_tokens": 74369846.0, |
| "reward": 0.3671875223517418, |
| "reward_std": 0.4153682075440884, |
| "rewards/reward_accuracy/mean": 0.2671875, |
| "rewards/reward_accuracy/std": 0.4153682075440884, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.595514714717865, |
| "sampling/importance_sampling_ratio/mean": 0.992092889547348, |
| "sampling/importance_sampling_ratio/min": 0.5417448043823242, |
| "sampling/sampling_logp_difference/max": 0.8134242266416549, |
| "sampling/sampling_logp_difference/mean": 0.0010209075175225736, |
| "step": 1350, |
| "step_time": 4.943212623638101 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 48.52265625, |
| "completions/mean_terminated_length": 48.52265625, |
| "completions/min_length": 24.9, |
| "completions/min_terminated_length": 24.9, |
| "entropy": 0.003395476038713241, |
| "epoch": 3.5142118863049094, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.864100000000001e-06, |
| "loss": -0.0053, |
| "num_tokens": 74912379.0, |
| "reward": 0.3976562723517418, |
| "reward_std": 0.44571022391319276, |
| "rewards/reward_accuracy/mean": 0.29765625, |
| "rewards/reward_accuracy/std": 0.44571022391319276, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3237088203430176, |
| "sampling/importance_sampling_ratio/mean": 0.9966547548770904, |
| "sampling/importance_sampling_ratio/min": 0.6519868716597557, |
| "sampling/sampling_logp_difference/max": 0.6258407397195697, |
| "sampling/sampling_logp_difference/mean": 0.0007848295848816633, |
| "step": 1360, |
| "step_time": 4.990737753664144 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 50.246875, |
| "completions/mean_terminated_length": 50.246875, |
| "completions/min_length": 27.6, |
| "completions/min_terminated_length": 27.6, |
| "entropy": 0.003960204482791596, |
| "epoch": 3.540051679586563, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 0.025634765625, |
| "learning_rate": 9.8631e-06, |
| "loss": -0.0088, |
| "num_tokens": 75459895.0, |
| "reward": 0.3992187738418579, |
| "reward_std": 0.4481558740139008, |
| "rewards/reward_accuracy/mean": 0.29921875, |
| "rewards/reward_accuracy/std": 0.4481558740139008, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5535614490509033, |
| "sampling/importance_sampling_ratio/mean": 0.998655903339386, |
| "sampling/importance_sampling_ratio/min": 0.5862275242805481, |
| "sampling/sampling_logp_difference/max": 0.63932312913239, |
| "sampling/sampling_logp_difference/mean": 0.000906339225548436, |
| "step": 1370, |
| "step_time": 4.902752069756389 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 50.10390625, |
| "completions/mean_terminated_length": 50.10390625, |
| "completions/min_length": 28.3, |
| "completions/min_terminated_length": 28.3, |
| "entropy": 0.003526580772268062, |
| "epoch": 3.565891472868217, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 1.1171875, |
| "learning_rate": 9.862100000000002e-06, |
| "loss": 0.0036, |
| "num_tokens": 76009036.0, |
| "reward": 0.39296876937150954, |
| "reward_std": 0.4347353219985962, |
| "rewards/reward_accuracy/mean": 0.29296875, |
| "rewards/reward_accuracy/std": 0.4347353160381317, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.57492755651474, |
| "sampling/importance_sampling_ratio/mean": 1.0037749588489533, |
| "sampling/importance_sampling_ratio/min": 0.5413332536816597, |
| "sampling/sampling_logp_difference/max": 0.8323452472686768, |
| "sampling/sampling_logp_difference/mean": 0.0008375843288376927, |
| "step": 1380, |
| "step_time": 4.968698855512775 |
| }, |
| { |
| "clip_ratio/high_max": 8.550808124709874e-05, |
| "clip_ratio/high_mean": 4.275404062354937e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.275404062354937e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 81.6, |
| "completions/max_terminated_length": 81.6, |
| "completions/mean_length": 49.7203125, |
| "completions/mean_terminated_length": 49.7203125, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.003179947379339865, |
| "epoch": 3.591731266149871, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.861100000000001e-06, |
| "loss": -0.0015, |
| "num_tokens": 76553966.0, |
| "reward": 0.3351562723517418, |
| "reward_std": 0.4186506479978561, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.4186506479978561, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.391466498374939, |
| "sampling/importance_sampling_ratio/mean": 1.0005066394805908, |
| "sampling/importance_sampling_ratio/min": 0.6362880542874336, |
| "sampling/sampling_logp_difference/max": 0.5635408341884613, |
| "sampling/sampling_logp_difference/mean": 0.0006224354321602732, |
| "step": 1390, |
| "step_time": 5.123214168753475 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 50.9828125, |
| "completions/mean_terminated_length": 50.9828125, |
| "completions/min_length": 28.9, |
| "completions/min_terminated_length": 28.9, |
| "entropy": 0.002675608095523785, |
| "epoch": 3.6175710594315245, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 1.2265625, |
| "learning_rate": 9.8601e-06, |
| "loss": -0.0026, |
| "num_tokens": 77104616.0, |
| "reward": 0.3679687693715096, |
| "reward_std": 0.4167696818709373, |
| "rewards/reward_accuracy/mean": 0.26796875, |
| "rewards/reward_accuracy/std": 0.4167696818709373, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3930428743362426, |
| "sampling/importance_sampling_ratio/mean": 0.9964837610721589, |
| "sampling/importance_sampling_ratio/min": 0.6281273052096367, |
| "sampling/sampling_logp_difference/max": 0.632700203359127, |
| "sampling/sampling_logp_difference/mean": 0.0007419579720590264, |
| "step": 1400, |
| "step_time": 4.950832971534692 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.3, |
| "completions/max_terminated_length": 73.3, |
| "completions/mean_length": 50.67109375, |
| "completions/mean_terminated_length": 50.67109375, |
| "completions/min_length": 23.5, |
| "completions/min_terminated_length": 23.5, |
| "entropy": 0.0030649593962152723, |
| "epoch": 3.6434108527131785, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8591e-06, |
| "loss": -0.0077, |
| "num_tokens": 77657091.0, |
| "reward": 0.4187500223517418, |
| "reward_std": 0.44533681869506836, |
| "rewards/reward_accuracy/mean": 0.31875, |
| "rewards/reward_accuracy/std": 0.44533681869506836, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7204724192619323, |
| "sampling/importance_sampling_ratio/mean": 0.998255443572998, |
| "sampling/importance_sampling_ratio/min": 0.5284227877855301, |
| "sampling/sampling_logp_difference/max": 0.87558713555336, |
| "sampling/sampling_logp_difference/mean": 0.0008013222890440374, |
| "step": 1410, |
| "step_time": 4.983909438550472 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 49.75234375, |
| "completions/mean_terminated_length": 49.75234375, |
| "completions/min_length": 25.8, |
| "completions/min_terminated_length": 25.8, |
| "entropy": 0.0025256051123051294, |
| "epoch": 3.669250645994832, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8581e-06, |
| "loss": 0.0027, |
| "num_tokens": 78204150.0, |
| "reward": 0.4390625223517418, |
| "reward_std": 0.4446574807167053, |
| "rewards/reward_accuracy/mean": 0.3390625, |
| "rewards/reward_accuracy/std": 0.4446574807167053, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3046709299087524, |
| "sampling/importance_sampling_ratio/mean": 1.0012662351131438, |
| "sampling/importance_sampling_ratio/min": 0.6212675452232361, |
| "sampling/sampling_logp_difference/max": 0.5605372648686171, |
| "sampling/sampling_logp_difference/mean": 0.0006228172978808288, |
| "step": 1420, |
| "step_time": 5.014857737859711 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.9, |
| "completions/max_terminated_length": 72.9, |
| "completions/mean_length": 49.640625, |
| "completions/mean_terminated_length": 49.640625, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.0021852816954378794, |
| "epoch": 3.695090439276486, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 1.5234375, |
| "learning_rate": 9.857100000000001e-06, |
| "loss": -0.0026, |
| "num_tokens": 78751546.0, |
| "reward": 0.4382812738418579, |
| "reward_std": 0.4594565451145172, |
| "rewards/reward_accuracy/mean": 0.33828125, |
| "rewards/reward_accuracy/std": 0.459456542134285, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5425957918167115, |
| "sampling/importance_sampling_ratio/mean": 1.0006016433238982, |
| "sampling/importance_sampling_ratio/min": 0.4848952040076256, |
| "sampling/sampling_logp_difference/max": 0.897838419675827, |
| "sampling/sampling_logp_difference/mean": 0.0007625755009939894, |
| "step": 1430, |
| "step_time": 4.963172565912828 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.0, |
| "completions/max_terminated_length": 72.0, |
| "completions/mean_length": 50.72109375, |
| "completions/mean_terminated_length": 50.72109375, |
| "completions/min_length": 27.8, |
| "completions/min_terminated_length": 27.8, |
| "entropy": 0.002017626014912821, |
| "epoch": 3.7209302325581395, |
| "frac_reward_zero_std": 0.99375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8561e-06, |
| "loss": 0.0009, |
| "num_tokens": 79301965.0, |
| "reward": 0.4257812723517418, |
| "reward_std": 0.45173735320568087, |
| "rewards/reward_accuracy/mean": 0.32578125, |
| "rewards/reward_accuracy/std": 0.45173735320568087, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2387989163398743, |
| "sampling/importance_sampling_ratio/mean": 1.001051551103592, |
| "sampling/importance_sampling_ratio/min": 0.8075616478919982, |
| "sampling/sampling_logp_difference/max": 0.3198650799691677, |
| "sampling/sampling_logp_difference/mean": 0.00032164359363378025, |
| "step": 1440, |
| "step_time": 4.913088271254674 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.4, |
| "completions/max_terminated_length": 70.4, |
| "completions/mean_length": 48.896875, |
| "completions/mean_terminated_length": 48.896875, |
| "completions/min_length": 26.9, |
| "completions/min_terminated_length": 26.9, |
| "entropy": 0.0031963614299456823, |
| "epoch": 3.746770025839793, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.855100000000002e-06, |
| "loss": -0.0011, |
| "num_tokens": 79846569.0, |
| "reward": 0.3476562723517418, |
| "reward_std": 0.41809754222631457, |
| "rewards/reward_accuracy/mean": 0.24765625, |
| "rewards/reward_accuracy/std": 0.4180975392460823, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6292582988739013, |
| "sampling/importance_sampling_ratio/mean": 0.9995667278766632, |
| "sampling/importance_sampling_ratio/min": 0.6069683119654655, |
| "sampling/sampling_logp_difference/max": 0.660994965583086, |
| "sampling/sampling_logp_difference/mean": 0.0007409185388496553, |
| "step": 1450, |
| "step_time": 4.945312909549102 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.9, |
| "completions/max_terminated_length": 74.9, |
| "completions/mean_length": 53.70078125, |
| "completions/mean_terminated_length": 53.70078125, |
| "completions/min_length": 27.5, |
| "completions/min_terminated_length": 27.5, |
| "entropy": 0.003139832189935987, |
| "epoch": 3.772609819121447, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 1.34375, |
| "learning_rate": 9.854100000000001e-06, |
| "loss": 0.0016, |
| "num_tokens": 80409978.0, |
| "reward": 0.36640626937150955, |
| "reward_std": 0.42211673557758334, |
| "rewards/reward_accuracy/mean": 0.26640625, |
| "rewards/reward_accuracy/std": 0.42211673557758334, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5562130451202392, |
| "sampling/importance_sampling_ratio/mean": 1.0016731083393098, |
| "sampling/importance_sampling_ratio/min": 0.6891585782170295, |
| "sampling/sampling_logp_difference/max": 0.6157727718353272, |
| "sampling/sampling_logp_difference/mean": 0.000653029493696522, |
| "step": 1460, |
| "step_time": 5.043970706686378 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.7, |
| "completions/max_terminated_length": 73.7, |
| "completions/mean_length": 50.12890625, |
| "completions/mean_terminated_length": 50.12890625, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.003870366747560183, |
| "epoch": 3.798449612403101, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 2.234375, |
| "learning_rate": 9.8531e-06, |
| "loss": -0.0092, |
| "num_tokens": 80957895.0, |
| "reward": 0.3734375238418579, |
| "reward_std": 0.43528908789157866, |
| "rewards/reward_accuracy/mean": 0.2734375, |
| "rewards/reward_accuracy/std": 0.43528908789157866, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7008383512496947, |
| "sampling/importance_sampling_ratio/mean": 1.001082068681717, |
| "sampling/importance_sampling_ratio/min": 0.6433152332901955, |
| "sampling/sampling_logp_difference/max": 0.7089414076879621, |
| "sampling/sampling_logp_difference/mean": 0.0007970935677803937, |
| "step": 1470, |
| "step_time": 4.952176067605615 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 77.2, |
| "completions/max_terminated_length": 77.2, |
| "completions/mean_length": 52.84140625, |
| "completions/mean_terminated_length": 52.84140625, |
| "completions/min_length": 27.8, |
| "completions/min_terminated_length": 27.8, |
| "entropy": 0.003655109567625914, |
| "epoch": 3.8242894056847545, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 1.0859375, |
| "learning_rate": 9.852100000000002e-06, |
| "loss": -0.0037, |
| "num_tokens": 81515884.0, |
| "reward": 0.41406252384185793, |
| "reward_std": 0.46062680184841154, |
| "rewards/reward_accuracy/mean": 0.3140625, |
| "rewards/reward_accuracy/std": 0.46062680184841154, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4946239590644836, |
| "sampling/importance_sampling_ratio/mean": 0.9949462711811066, |
| "sampling/importance_sampling_ratio/min": 0.5154244437813759, |
| "sampling/sampling_logp_difference/max": 0.5613792836666107, |
| "sampling/sampling_logp_difference/mean": 0.0009483149216976017, |
| "step": 1480, |
| "step_time": 5.0573271457804365 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 48.85546875, |
| "completions/mean_terminated_length": 48.85546875, |
| "completions/min_length": 28.6, |
| "completions/min_terminated_length": 28.6, |
| "entropy": 0.0028006504120639875, |
| "epoch": 3.850129198966408, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.851100000000001e-06, |
| "loss": -0.001, |
| "num_tokens": 82058907.0, |
| "reward": 0.42578127384185793, |
| "reward_std": 0.458673033118248, |
| "rewards/reward_accuracy/mean": 0.32578125, |
| "rewards/reward_accuracy/std": 0.458673033118248, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5501605868339539, |
| "sampling/importance_sampling_ratio/mean": 0.9957859337329864, |
| "sampling/importance_sampling_ratio/min": 0.5988857418298721, |
| "sampling/sampling_logp_difference/max": 0.7202365385368467, |
| "sampling/sampling_logp_difference/mean": 0.0010242652359011117, |
| "step": 1490, |
| "step_time": 5.00651352070272 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.4, |
| "completions/max_terminated_length": 71.4, |
| "completions/mean_length": 48.83125, |
| "completions/mean_terminated_length": 48.83125, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.0022598848346206068, |
| "epoch": 3.875968992248062, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 4.3125, |
| "learning_rate": 9.8501e-06, |
| "loss": -0.0046, |
| "num_tokens": 82602795.0, |
| "reward": 0.3937500223517418, |
| "reward_std": 0.43583558648824694, |
| "rewards/reward_accuracy/mean": 0.29375, |
| "rewards/reward_accuracy/std": 0.43583558648824694, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3795061349868774, |
| "sampling/importance_sampling_ratio/mean": 0.9907485365867614, |
| "sampling/importance_sampling_ratio/min": 0.43262876719236376, |
| "sampling/sampling_logp_difference/max": 0.9324224889278412, |
| "sampling/sampling_logp_difference/mean": 0.0007077434660459403, |
| "step": 1500, |
| "step_time": 4.969147672853433 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.9752592672593893e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.9752592672593893e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.5, |
| "completions/max_terminated_length": 72.5, |
| "completions/mean_length": 50.0953125, |
| "completions/mean_terminated_length": 50.0953125, |
| "completions/min_length": 26.7, |
| "completions/min_terminated_length": 26.7, |
| "entropy": 0.002711671697124984, |
| "epoch": 3.901808785529716, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 2.796875, |
| "learning_rate": 9.8491e-06, |
| "loss": 0.0021, |
| "num_tokens": 83151557.0, |
| "reward": 0.42031252235174177, |
| "reward_std": 0.45238761603832245, |
| "rewards/reward_accuracy/mean": 0.3203125, |
| "rewards/reward_accuracy/std": 0.45238761603832245, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4638926148414613, |
| "sampling/importance_sampling_ratio/mean": 0.9980453312397003, |
| "sampling/importance_sampling_ratio/min": 0.6400665365159511, |
| "sampling/sampling_logp_difference/max": 0.6958723668009043, |
| "sampling/sampling_logp_difference/mean": 0.000732313999105827, |
| "step": 1510, |
| "step_time": 5.010396922519431 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.9, |
| "completions/max_terminated_length": 73.9, |
| "completions/mean_length": 49.978125, |
| "completions/mean_terminated_length": 49.978125, |
| "completions/min_length": 29.1, |
| "completions/min_terminated_length": 29.1, |
| "entropy": 0.0030437848843575923, |
| "epoch": 3.9276485788113695, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8481e-06, |
| "loss": -0.0, |
| "num_tokens": 83698977.0, |
| "reward": 0.37265627086162567, |
| "reward_std": 0.4326770007610321, |
| "rewards/reward_accuracy/mean": 0.27265625, |
| "rewards/reward_accuracy/std": 0.4326770007610321, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5197882175445556, |
| "sampling/importance_sampling_ratio/mean": 0.9986387908458709, |
| "sampling/importance_sampling_ratio/min": 0.6784495577216149, |
| "sampling/sampling_logp_difference/max": 0.6796332836151123, |
| "sampling/sampling_logp_difference/mean": 0.00047000525664770977, |
| "step": 1520, |
| "step_time": 4.92452270637732 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.6, |
| "completions/max_terminated_length": 71.6, |
| "completions/mean_length": 50.68359375, |
| "completions/mean_terminated_length": 50.68359375, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.003609791256440076, |
| "epoch": 3.953488372093023, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 1.8359375, |
| "learning_rate": 9.847100000000001e-06, |
| "loss": 0.0017, |
| "num_tokens": 84250420.0, |
| "reward": 0.3671875223517418, |
| "reward_std": 0.4215258464217186, |
| "rewards/reward_accuracy/mean": 0.2671875, |
| "rewards/reward_accuracy/std": 0.4215258464217186, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8333860516548157, |
| "sampling/importance_sampling_ratio/mean": 1.0089458525180817, |
| "sampling/importance_sampling_ratio/min": 0.49295970648527143, |
| "sampling/sampling_logp_difference/max": 0.8971081078052521, |
| "sampling/sampling_logp_difference/mean": 0.0010437624587211758, |
| "step": 1530, |
| "step_time": 5.037635967088863 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 51.1609375, |
| "completions/mean_terminated_length": 51.1609375, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.0036470654358709, |
| "epoch": 3.9793281653746773, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8461e-06, |
| "loss": -0.0004, |
| "num_tokens": 84802682.0, |
| "reward": 0.3546875208616257, |
| "reward_std": 0.41496777534484863, |
| "rewards/reward_accuracy/mean": 0.2546875, |
| "rewards/reward_accuracy/std": 0.41496777534484863, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7330074787139893, |
| "sampling/importance_sampling_ratio/mean": 0.9984236776828765, |
| "sampling/importance_sampling_ratio/min": 0.47876126170158384, |
| "sampling/sampling_logp_difference/max": 0.8474727034568786, |
| "sampling/sampling_logp_difference/mean": 0.000904385483590886, |
| "step": 1540, |
| "step_time": 4.990532020409591 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.2, |
| "completions/max_terminated_length": 72.2, |
| "completions/mean_length": 47.0875, |
| "completions/mean_terminated_length": 47.0875, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.002953325026010134, |
| "epoch": 4.0051679586563305, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.3359375, |
| "learning_rate": 9.8451e-06, |
| "loss": -0.0017, |
| "num_tokens": 85340298.0, |
| "reward": 0.48828127384185793, |
| "reward_std": 0.47929089665412905, |
| "rewards/reward_accuracy/mean": 0.38828125, |
| "rewards/reward_accuracy/std": 0.47929089665412905, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4553777575492859, |
| "sampling/importance_sampling_ratio/mean": 0.99609255194664, |
| "sampling/importance_sampling_ratio/min": 0.6326152563095093, |
| "sampling/sampling_logp_difference/max": 0.6558850526809692, |
| "sampling/sampling_logp_difference/mean": 0.0006759868301742245, |
| "step": 1550, |
| "step_time": 4.940871482458897 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 48.57578125, |
| "completions/mean_terminated_length": 48.57578125, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.0034923404191431473, |
| "epoch": 4.0310077519379846, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.844100000000001e-06, |
| "loss": -0.0056, |
| "num_tokens": 85883139.0, |
| "reward": 0.42968752384185793, |
| "reward_std": 0.45426276326179504, |
| "rewards/reward_accuracy/mean": 0.3296875, |
| "rewards/reward_accuracy/std": 0.45426276326179504, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5458737015724182, |
| "sampling/importance_sampling_ratio/mean": 1.0002525746822357, |
| "sampling/importance_sampling_ratio/min": 0.6412003070116044, |
| "sampling/sampling_logp_difference/max": 0.7055759847164154, |
| "sampling/sampling_logp_difference/mean": 0.000949995216797106, |
| "step": 1560, |
| "step_time": 4.963002066593617 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.9, |
| "completions/max_terminated_length": 71.9, |
| "completions/mean_length": 49.54140625, |
| "completions/mean_terminated_length": 49.54140625, |
| "completions/min_length": 28.8, |
| "completions/min_terminated_length": 28.8, |
| "entropy": 0.0031189505736620047, |
| "epoch": 4.056847545219639, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8431e-06, |
| "loss": 0.0034, |
| "num_tokens": 86429848.0, |
| "reward": 0.4203125238418579, |
| "reward_std": 0.45688324570655825, |
| "rewards/reward_accuracy/mean": 0.3203125, |
| "rewards/reward_accuracy/std": 0.45688324570655825, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3193398237228393, |
| "sampling/importance_sampling_ratio/mean": 0.9921945631504059, |
| "sampling/importance_sampling_ratio/min": 0.545482975244522, |
| "sampling/sampling_logp_difference/max": 0.6889259338378906, |
| "sampling/sampling_logp_difference/mean": 0.000683220915379934, |
| "step": 1570, |
| "step_time": 4.954102780367248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.9, |
| "completions/max_terminated_length": 71.9, |
| "completions/mean_length": 49.859375, |
| "completions/mean_terminated_length": 49.859375, |
| "completions/min_length": 27.0, |
| "completions/min_terminated_length": 27.0, |
| "entropy": 0.002925300772039918, |
| "epoch": 4.082687338501292, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.842100000000002e-06, |
| "loss": 0.0007, |
| "num_tokens": 86976492.0, |
| "reward": 0.4054687723517418, |
| "reward_std": 0.44635605812072754, |
| "rewards/reward_accuracy/mean": 0.30546875, |
| "rewards/reward_accuracy/std": 0.44635605812072754, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6135616779327393, |
| "sampling/importance_sampling_ratio/mean": 0.9971276104450226, |
| "sampling/importance_sampling_ratio/min": 0.592396256327629, |
| "sampling/sampling_logp_difference/max": 0.6663916885852814, |
| "sampling/sampling_logp_difference/mean": 0.0007383194373687729, |
| "step": 1580, |
| "step_time": 4.855978901800699 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.4, |
| "completions/max_terminated_length": 73.4, |
| "completions/mean_length": 50.0390625, |
| "completions/mean_terminated_length": 50.0390625, |
| "completions/min_length": 27.2, |
| "completions/min_terminated_length": 27.2, |
| "entropy": 0.0032671983183718114, |
| "epoch": 4.108527131782946, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.841100000000001e-06, |
| "loss": -0.0003, |
| "num_tokens": 87524870.0, |
| "reward": 0.41406252384185793, |
| "reward_std": 0.45002243518829343, |
| "rewards/reward_accuracy/mean": 0.3140625, |
| "rewards/reward_accuracy/std": 0.4500224322080612, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4601199269294738, |
| "sampling/importance_sampling_ratio/mean": 0.9982122778892517, |
| "sampling/importance_sampling_ratio/min": 0.6432103663682938, |
| "sampling/sampling_logp_difference/max": 0.5924935936927795, |
| "sampling/sampling_logp_difference/mean": 0.0006280679750489071, |
| "step": 1590, |
| "step_time": 5.040452429605648 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.4, |
| "completions/max_terminated_length": 72.4, |
| "completions/mean_length": 49.20546875, |
| "completions/mean_terminated_length": 49.20546875, |
| "completions/min_length": 25.0, |
| "completions/min_terminated_length": 25.0, |
| "entropy": 0.003809290586013958, |
| "epoch": 4.134366925064599, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.840100000000001e-06, |
| "loss": 0.0056, |
| "num_tokens": 88068397.0, |
| "reward": 0.3546875223517418, |
| "reward_std": 0.4236186847090721, |
| "rewards/reward_accuracy/mean": 0.2546875, |
| "rewards/reward_accuracy/std": 0.4236186847090721, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6732622504234314, |
| "sampling/importance_sampling_ratio/mean": 1.0034389793872833, |
| "sampling/importance_sampling_ratio/min": 0.5605317413806915, |
| "sampling/sampling_logp_difference/max": 0.6425735712051391, |
| "sampling/sampling_logp_difference/mean": 0.0007306865241844207, |
| "step": 1600, |
| "step_time": 5.002290871390142 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.5, |
| "completions/max_terminated_length": 73.5, |
| "completions/mean_length": 50.6859375, |
| "completions/mean_terminated_length": 50.6859375, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.0045609254450027946, |
| "epoch": 4.160206718346253, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.7109375, |
| "learning_rate": 9.8391e-06, |
| "loss": -0.0031, |
| "num_tokens": 88619699.0, |
| "reward": 0.4273437723517418, |
| "reward_std": 0.45399353504180906, |
| "rewards/reward_accuracy/mean": 0.32734375, |
| "rewards/reward_accuracy/std": 0.45399353504180906, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.626962959766388, |
| "sampling/importance_sampling_ratio/mean": 1.0007759869098662, |
| "sampling/importance_sampling_ratio/min": 0.543695256114006, |
| "sampling/sampling_logp_difference/max": 0.709826922416687, |
| "sampling/sampling_logp_difference/mean": 0.0009675839683040977, |
| "step": 1610, |
| "step_time": 4.94715522469487 |
| }, |
| { |
| "clip_ratio/high_max": 5.56032289750874e-05, |
| "clip_ratio/high_mean": 2.78016144875437e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.78016144875437e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.6, |
| "completions/max_terminated_length": 73.6, |
| "completions/mean_length": 51.4890625, |
| "completions/mean_terminated_length": 51.4890625, |
| "completions/min_length": 26.2, |
| "completions/min_terminated_length": 26.2, |
| "entropy": 0.005426073006856313, |
| "epoch": 4.186046511627907, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8381e-06, |
| "loss": -0.0001, |
| "num_tokens": 89173341.0, |
| "reward": 0.3882812738418579, |
| "reward_std": 0.44585159718990325, |
| "rewards/reward_accuracy/mean": 0.28828125, |
| "rewards/reward_accuracy/std": 0.44585159718990325, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.503739333152771, |
| "sampling/importance_sampling_ratio/mean": 0.9979479789733887, |
| "sampling/importance_sampling_ratio/min": 0.5694765210151672, |
| "sampling/sampling_logp_difference/max": 0.6299612045288085, |
| "sampling/sampling_logp_difference/mean": 0.0008188476029317826, |
| "step": 1620, |
| "step_time": 4.9979641838697715 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.7, |
| "completions/max_terminated_length": 74.7, |
| "completions/mean_length": 50.46015625, |
| "completions/mean_terminated_length": 50.46015625, |
| "completions/min_length": 28.0, |
| "completions/min_terminated_length": 28.0, |
| "entropy": 0.007376608792765182, |
| "epoch": 4.2118863049095605, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.271484375, |
| "learning_rate": 9.837100000000001e-06, |
| "loss": 0.0032, |
| "num_tokens": 89724754.0, |
| "reward": 0.4226562723517418, |
| "reward_std": 0.4501898318529129, |
| "rewards/reward_accuracy/mean": 0.32265625, |
| "rewards/reward_accuracy/std": 0.4501898318529129, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5044286608695985, |
| "sampling/importance_sampling_ratio/mean": 0.9954514026641845, |
| "sampling/importance_sampling_ratio/min": 0.5263983085751534, |
| "sampling/sampling_logp_difference/max": 0.7020433068275451, |
| "sampling/sampling_logp_difference/mean": 0.001168558862991631, |
| "step": 1630, |
| "step_time": 5.02375446909573 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.1, |
| "completions/max_terminated_length": 73.1, |
| "completions/mean_length": 50.109375, |
| "completions/mean_terminated_length": 50.109375, |
| "completions/min_length": 26.5, |
| "completions/min_terminated_length": 26.5, |
| "entropy": 0.009699543903843732, |
| "epoch": 4.237726098191215, |
| "frac_reward_zero_std": 0.90625, |
| "grad_norm": 1.1640625, |
| "learning_rate": 9.8361e-06, |
| "loss": 0.0071, |
| "num_tokens": 90272846.0, |
| "reward": 0.4890625238418579, |
| "reward_std": 0.4591934770345688, |
| "rewards/reward_accuracy/mean": 0.3890625, |
| "rewards/reward_accuracy/std": 0.4591934770345688, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6726544618606567, |
| "sampling/importance_sampling_ratio/mean": 0.9972900211811065, |
| "sampling/importance_sampling_ratio/min": 0.5256831780076027, |
| "sampling/sampling_logp_difference/max": 0.6736291527748108, |
| "sampling/sampling_logp_difference/mean": 0.0014150463859550655, |
| "step": 1640, |
| "step_time": 5.0543996474007145 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.0, |
| "completions/max_terminated_length": 74.0, |
| "completions/mean_length": 51.7984375, |
| "completions/mean_terminated_length": 51.7984375, |
| "completions/min_length": 26.9, |
| "completions/min_terminated_length": 26.9, |
| "entropy": 0.011362850326986518, |
| "epoch": 4.263565891472869, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.67578125, |
| "learning_rate": 9.8351e-06, |
| "loss": 0.0006, |
| "num_tokens": 90827124.0, |
| "reward": 0.3460156500339508, |
| "reward_std": 0.4251233696937561, |
| "rewards/reward_accuracy/mean": 0.24609375, |
| "rewards/reward_accuracy/std": 0.4250654250383377, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006225345656275749, |
| "sampling/importance_sampling_ratio/max": 1.540428066253662, |
| "sampling/importance_sampling_ratio/mean": 0.9918396294116973, |
| "sampling/importance_sampling_ratio/min": 0.47347278594970704, |
| "sampling/sampling_logp_difference/max": 0.9509158372879029, |
| "sampling/sampling_logp_difference/mean": 0.001620137330610305, |
| "step": 1650, |
| "step_time": 4.990306184743531 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 75.1, |
| "completions/max_terminated_length": 75.1, |
| "completions/mean_length": 50.62421875, |
| "completions/mean_terminated_length": 50.62421875, |
| "completions/min_length": 26.9, |
| "completions/min_terminated_length": 26.9, |
| "entropy": 0.010891792371694464, |
| "epoch": 4.289405684754522, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 1.4375, |
| "learning_rate": 9.834100000000001e-06, |
| "loss": -0.0073, |
| "num_tokens": 91377427.0, |
| "reward": 0.40386721193790437, |
| "reward_std": 0.460090571641922, |
| "rewards/reward_accuracy/mean": 0.30390625, |
| "rewards/reward_accuracy/std": 0.4600625067949295, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.6204795002937318, |
| "sampling/importance_sampling_ratio/mean": 0.9972130417823791, |
| "sampling/importance_sampling_ratio/min": 0.5764504820108414, |
| "sampling/sampling_logp_difference/max": 0.6256880044937134, |
| "sampling/sampling_logp_difference/mean": 0.0014447858760831878, |
| "step": 1660, |
| "step_time": 5.012071219389327 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 85.5, |
| "completions/max_terminated_length": 85.5, |
| "completions/mean_length": 51.36171875, |
| "completions/mean_terminated_length": 51.36171875, |
| "completions/min_length": 23.9, |
| "completions/min_terminated_length": 23.9, |
| "entropy": 0.007680761791925761, |
| "epoch": 4.315245478036176, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.35546875, |
| "learning_rate": 9.8331e-06, |
| "loss": 0.0027, |
| "num_tokens": 91930418.0, |
| "reward": 0.4046875238418579, |
| "reward_std": 0.45459361672401427, |
| "rewards/reward_accuracy/mean": 0.3046875, |
| "rewards/reward_accuracy/std": 0.45459361672401427, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.469143545627594, |
| "sampling/importance_sampling_ratio/mean": 0.9885624051094055, |
| "sampling/importance_sampling_ratio/min": 0.3741734802722931, |
| "sampling/sampling_logp_difference/max": 1.0310903251171113, |
| "sampling/sampling_logp_difference/mean": 0.0014649431046564131, |
| "step": 1670, |
| "step_time": 5.1247962295077745 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 77.6, |
| "completions/max_terminated_length": 77.6, |
| "completions/mean_length": 50.984375, |
| "completions/mean_terminated_length": 50.984375, |
| "completions/min_length": 28.2, |
| "completions/min_terminated_length": 28.2, |
| "entropy": 0.005211545509519055, |
| "epoch": 4.341085271317829, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8321e-06, |
| "loss": -0.0043, |
| "num_tokens": 92482974.0, |
| "reward": 0.39843752384185793, |
| "reward_std": 0.45071819722652434, |
| "rewards/reward_accuracy/mean": 0.2984375, |
| "rewards/reward_accuracy/std": 0.45071819722652434, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5405234336853026, |
| "sampling/importance_sampling_ratio/mean": 0.9944289147853851, |
| "sampling/importance_sampling_ratio/min": 0.36753590404987335, |
| "sampling/sampling_logp_difference/max": 1.2192301511764527, |
| "sampling/sampling_logp_difference/mean": 0.001350799432839267, |
| "step": 1680, |
| "step_time": 5.099342879932374 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 50.82734375, |
| "completions/mean_terminated_length": 50.82734375, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.005230863075667003, |
| "epoch": 4.366925064599483, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 4.15625, |
| "learning_rate": 9.831100000000001e-06, |
| "loss": -0.0039, |
| "num_tokens": 93034857.0, |
| "reward": 0.35390627235174177, |
| "reward_std": 0.424275267124176, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4242752641439438, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5476924300193786, |
| "sampling/importance_sampling_ratio/mean": 0.9983605921268464, |
| "sampling/importance_sampling_ratio/min": 0.483190244436264, |
| "sampling/sampling_logp_difference/max": 0.850607055425644, |
| "sampling/sampling_logp_difference/mean": 0.0011433307692641393, |
| "step": 1690, |
| "step_time": 4.966709339129738 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 77.9, |
| "completions/max_terminated_length": 77.9, |
| "completions/mean_length": 50.7390625, |
| "completions/mean_terminated_length": 50.7390625, |
| "completions/min_length": 24.3, |
| "completions/min_terminated_length": 24.3, |
| "entropy": 0.005109614793946093, |
| "epoch": 4.392764857881137, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.5390625, |
| "learning_rate": 9.830100000000001e-06, |
| "loss": -0.0034, |
| "num_tokens": 93584339.0, |
| "reward": 0.39687501937150954, |
| "reward_std": 0.4253468945622444, |
| "rewards/reward_accuracy/mean": 0.296875, |
| "rewards/reward_accuracy/std": 0.4253468945622444, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4950310230255126, |
| "sampling/importance_sampling_ratio/mean": 0.9953265905380249, |
| "sampling/importance_sampling_ratio/min": 0.5447196036577224, |
| "sampling/sampling_logp_difference/max": 0.7453190946951509, |
| "sampling/sampling_logp_difference/mean": 0.0009144828916760161, |
| "step": 1700, |
| "step_time": 5.031076519773341 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012083971814718098, |
| "clip_ratio/high_mean": 6.041985907359049e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.041985907359049e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.6, |
| "completions/max_terminated_length": 74.6, |
| "completions/mean_length": 50.86640625, |
| "completions/mean_terminated_length": 50.86640625, |
| "completions/min_length": 27.0, |
| "completions/min_terminated_length": 27.0, |
| "entropy": 0.003196481067698187, |
| "epoch": 4.4186046511627906, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8291e-06, |
| "loss": 0.0002, |
| "num_tokens": 94136424.0, |
| "reward": 0.3406250238418579, |
| "reward_std": 0.4227444350719452, |
| "rewards/reward_accuracy/mean": 0.240625, |
| "rewards/reward_accuracy/std": 0.4227444350719452, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5126087307929992, |
| "sampling/importance_sampling_ratio/mean": 0.9933084785938263, |
| "sampling/importance_sampling_ratio/min": 0.43274506032466886, |
| "sampling/sampling_logp_difference/max": 0.9141636788845062, |
| "sampling/sampling_logp_difference/mean": 0.000994291339884512, |
| "step": 1710, |
| "step_time": 4.987809217884205 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 98.9, |
| "completions/max_terminated_length": 98.9, |
| "completions/mean_length": 51.0109375, |
| "completions/mean_terminated_length": 51.0109375, |
| "completions/min_length": 24.8, |
| "completions/min_terminated_length": 24.8, |
| "entropy": 0.00549568508995435, |
| "epoch": 4.444444444444445, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 1.2734375, |
| "learning_rate": 9.8281e-06, |
| "loss": -0.0024, |
| "num_tokens": 94689102.0, |
| "reward": 0.3437500223517418, |
| "reward_std": 0.4216102480888367, |
| "rewards/reward_accuracy/mean": 0.24375, |
| "rewards/reward_accuracy/std": 0.4216102480888367, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6126699686050414, |
| "sampling/importance_sampling_ratio/mean": 0.9918075144290924, |
| "sampling/importance_sampling_ratio/min": 0.3134855590760708, |
| "sampling/sampling_logp_difference/max": 1.068973296880722, |
| "sampling/sampling_logp_difference/mean": 0.0017597652622498572, |
| "step": 1720, |
| "step_time": 5.431200831918977 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 52.38515625, |
| "completions/mean_terminated_length": 52.38515625, |
| "completions/min_length": 28.6, |
| "completions/min_terminated_length": 28.6, |
| "entropy": 0.0030088349828474746, |
| "epoch": 4.470284237726098, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.827100000000001e-06, |
| "loss": -0.0061, |
| "num_tokens": 95246219.0, |
| "reward": 0.39687502235174177, |
| "reward_std": 0.4289079800248146, |
| "rewards/reward_accuracy/mean": 0.296875, |
| "rewards/reward_accuracy/std": 0.4289079800248146, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7307371735572814, |
| "sampling/importance_sampling_ratio/mean": 0.999963766336441, |
| "sampling/importance_sampling_ratio/min": 0.624581691622734, |
| "sampling/sampling_logp_difference/max": 0.6656971631571651, |
| "sampling/sampling_logp_difference/mean": 0.0008267551131211804, |
| "step": 1730, |
| "step_time": 4.906079669483006 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 130.8, |
| "completions/max_terminated_length": 130.8, |
| "completions/mean_length": 49.965625, |
| "completions/mean_terminated_length": 49.965625, |
| "completions/min_length": 25.7, |
| "completions/min_terminated_length": 25.7, |
| "entropy": 0.006619979083734506, |
| "epoch": 4.496124031007752, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8261e-06, |
| "loss": -0.0013, |
| "num_tokens": 95790791.0, |
| "reward": 0.406992207467556, |
| "reward_std": 0.4395966172218323, |
| "rewards/reward_accuracy/mean": 0.30703125, |
| "rewards/reward_accuracy/std": 0.4395509123802185, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.468680214881897, |
| "sampling/importance_sampling_ratio/mean": 0.9963645517826081, |
| "sampling/importance_sampling_ratio/min": 0.5971707701683044, |
| "sampling/sampling_logp_difference/max": 0.5401690840721131, |
| "sampling/sampling_logp_difference/mean": 0.0010511542452150025, |
| "step": 1740, |
| "step_time": 5.786183668347076 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.9, |
| "completions/max_terminated_length": 71.9, |
| "completions/mean_length": 50.65234375, |
| "completions/mean_terminated_length": 50.65234375, |
| "completions/min_length": 29.1, |
| "completions/min_terminated_length": 29.1, |
| "entropy": 0.0035814286436107066, |
| "epoch": 4.521963824289406, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8251e-06, |
| "loss": -0.0025, |
| "num_tokens": 96342154.0, |
| "reward": 0.4015625223517418, |
| "reward_std": 0.44325951039791106, |
| "rewards/reward_accuracy/mean": 0.3015625, |
| "rewards/reward_accuracy/std": 0.44325951039791106, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8498161554336547, |
| "sampling/importance_sampling_ratio/mean": 0.9945340991020203, |
| "sampling/importance_sampling_ratio/min": 0.5848314106464386, |
| "sampling/sampling_logp_difference/max": 0.8216823935508728, |
| "sampling/sampling_logp_difference/mean": 0.0010463000216986984, |
| "step": 1750, |
| "step_time": 4.931607118854299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 49.23671875, |
| "completions/mean_terminated_length": 49.23671875, |
| "completions/min_length": 27.0, |
| "completions/min_terminated_length": 27.0, |
| "entropy": 0.004031703076543635, |
| "epoch": 4.547803617571059, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.824100000000001e-06, |
| "loss": -0.0002, |
| "num_tokens": 96888161.0, |
| "reward": 0.4000000238418579, |
| "reward_std": 0.4488473415374756, |
| "rewards/reward_accuracy/mean": 0.3, |
| "rewards/reward_accuracy/std": 0.4488473415374756, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4037821769714356, |
| "sampling/importance_sampling_ratio/mean": 0.9979808032512665, |
| "sampling/importance_sampling_ratio/min": 0.6709481000900268, |
| "sampling/sampling_logp_difference/max": 0.507661247253418, |
| "sampling/sampling_logp_difference/mean": 0.000726495633716695, |
| "step": 1760, |
| "step_time": 4.918747498723678 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 49.58515625, |
| "completions/mean_terminated_length": 49.58515625, |
| "completions/min_length": 25.4, |
| "completions/min_terminated_length": 25.4, |
| "entropy": 0.0050822914228774605, |
| "epoch": 4.573643410852713, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.823100000000001e-06, |
| "loss": 0.0009, |
| "num_tokens": 97435398.0, |
| "reward": 0.4000000238418579, |
| "reward_std": 0.446417036652565, |
| "rewards/reward_accuracy/mean": 0.3, |
| "rewards/reward_accuracy/std": 0.446417036652565, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5059492468833924, |
| "sampling/importance_sampling_ratio/mean": 0.993720817565918, |
| "sampling/importance_sampling_ratio/min": 0.5610156953334808, |
| "sampling/sampling_logp_difference/max": 0.6258687794208526, |
| "sampling/sampling_logp_difference/mean": 0.0008405594562646002, |
| "step": 1770, |
| "step_time": 5.018867579801008 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.6, |
| "completions/max_terminated_length": 71.6, |
| "completions/mean_length": 49.11796875, |
| "completions/mean_terminated_length": 49.11796875, |
| "completions/min_length": 26.7, |
| "completions/min_terminated_length": 26.7, |
| "entropy": 0.005187274546005938, |
| "epoch": 4.599483204134367, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.98046875, |
| "learning_rate": 9.8221e-06, |
| "loss": 0.0006, |
| "num_tokens": 97980189.0, |
| "reward": 0.4117187738418579, |
| "reward_std": 0.44791279435157777, |
| "rewards/reward_accuracy/mean": 0.31171875, |
| "rewards/reward_accuracy/std": 0.44791279137134554, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5372934937477112, |
| "sampling/importance_sampling_ratio/mean": 0.9958673298358918, |
| "sampling/importance_sampling_ratio/min": 0.5076551169157029, |
| "sampling/sampling_logp_difference/max": 0.7294674336910247, |
| "sampling/sampling_logp_difference/mean": 0.0010391735762823374, |
| "step": 1780, |
| "step_time": 4.869200729671865 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.4, |
| "completions/max_terminated_length": 74.4, |
| "completions/mean_length": 50.90625, |
| "completions/mean_terminated_length": 50.90625, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.005822549397998955, |
| "epoch": 4.625322997416021, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.821100000000002e-06, |
| "loss": -0.0026, |
| "num_tokens": 98531957.0, |
| "reward": 0.39062502384185793, |
| "reward_std": 0.44793511629104615, |
| "rewards/reward_accuracy/mean": 0.290625, |
| "rewards/reward_accuracy/std": 0.44793511629104615, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.849425494670868, |
| "sampling/importance_sampling_ratio/mean": 1.0028749704360962, |
| "sampling/importance_sampling_ratio/min": 0.5931536518037319, |
| "sampling/sampling_logp_difference/max": 0.878334105014801, |
| "sampling/sampling_logp_difference/mean": 0.0010628742165863514, |
| "step": 1790, |
| "step_time": 5.064653449715115 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.0, |
| "completions/max_terminated_length": 73.0, |
| "completions/mean_length": 50.3921875, |
| "completions/mean_terminated_length": 50.3921875, |
| "completions/min_length": 27.0, |
| "completions/min_terminated_length": 27.0, |
| "entropy": 0.006874338424677262, |
| "epoch": 4.651162790697675, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 2.734375, |
| "learning_rate": 9.820100000000001e-06, |
| "loss": -0.0072, |
| "num_tokens": 99080907.0, |
| "reward": 0.4562500238418579, |
| "reward_std": 0.47185078561306, |
| "rewards/reward_accuracy/mean": 0.35625, |
| "rewards/reward_accuracy/std": 0.47185078561306, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.677233350276947, |
| "sampling/importance_sampling_ratio/mean": 1.0009793579578399, |
| "sampling/importance_sampling_ratio/min": 0.608161211013794, |
| "sampling/sampling_logp_difference/max": 0.6484045207500457, |
| "sampling/sampling_logp_difference/mean": 0.001090289742569439, |
| "step": 1800, |
| "step_time": 4.961302284919657 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.4, |
| "completions/max_terminated_length": 73.4, |
| "completions/mean_length": 50.021875, |
| "completions/mean_terminated_length": 50.021875, |
| "completions/min_length": 26.8, |
| "completions/min_terminated_length": 26.8, |
| "entropy": 0.005163963167069597, |
| "epoch": 4.677002583979328, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 1.484375, |
| "learning_rate": 9.8191e-06, |
| "loss": 0.0016, |
| "num_tokens": 99628375.0, |
| "reward": 0.4382812723517418, |
| "reward_std": 0.4531688064336777, |
| "rewards/reward_accuracy/mean": 0.33828125, |
| "rewards/reward_accuracy/std": 0.4531688064336777, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7652442574501037, |
| "sampling/importance_sampling_ratio/mean": 1.002433979511261, |
| "sampling/importance_sampling_ratio/min": 0.6260219484567642, |
| "sampling/sampling_logp_difference/max": 0.6745835483074188, |
| "sampling/sampling_logp_difference/mean": 0.0010315612074919046, |
| "step": 1810, |
| "step_time": 4.937012254493311 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 50.77109375, |
| "completions/mean_terminated_length": 50.77109375, |
| "completions/min_length": 25.5, |
| "completions/min_terminated_length": 25.5, |
| "entropy": 0.005661701964345412, |
| "epoch": 4.702842377260982, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8181e-06, |
| "loss": 0.0026, |
| "num_tokens": 100180978.0, |
| "reward": 0.4710937738418579, |
| "reward_std": 0.46909780204296114, |
| "rewards/reward_accuracy/mean": 0.37109375, |
| "rewards/reward_accuracy/std": 0.46909780204296114, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4520383477210999, |
| "sampling/importance_sampling_ratio/mean": 0.9968438625335694, |
| "sampling/importance_sampling_ratio/min": 0.5477475211024284, |
| "sampling/sampling_logp_difference/max": 0.7780803322792054, |
| "sampling/sampling_logp_difference/mean": 0.0009923454432282596, |
| "step": 1820, |
| "step_time": 5.083134080399759 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.1, |
| "completions/max_terminated_length": 72.1, |
| "completions/mean_length": 50.5859375, |
| "completions/mean_terminated_length": 50.5859375, |
| "completions/min_length": 23.5, |
| "completions/min_terminated_length": 23.5, |
| "entropy": 0.004795131613172998, |
| "epoch": 4.728682170542635, |
| "frac_reward_zero_std": 0.9875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.817100000000001e-06, |
| "loss": 0.0005, |
| "num_tokens": 100730880.0, |
| "reward": 0.3562500223517418, |
| "reward_std": 0.4245426312088966, |
| "rewards/reward_accuracy/mean": 0.25625, |
| "rewards/reward_accuracy/std": 0.4245426312088966, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7454367399215698, |
| "sampling/importance_sampling_ratio/mean": 0.9980115771293641, |
| "sampling/importance_sampling_ratio/min": 0.4807202696800232, |
| "sampling/sampling_logp_difference/max": 0.9421596646308898, |
| "sampling/sampling_logp_difference/mean": 0.0010480585042387247, |
| "step": 1830, |
| "step_time": 4.862088537588716 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.9, |
| "completions/max_terminated_length": 72.9, |
| "completions/mean_length": 51.38359375, |
| "completions/mean_terminated_length": 51.38359375, |
| "completions/min_length": 27.3, |
| "completions/min_terminated_length": 27.3, |
| "entropy": 0.005762942194996868, |
| "epoch": 4.754521963824289, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8161e-06, |
| "loss": 0.0028, |
| "num_tokens": 101283491.0, |
| "reward": 0.4039062738418579, |
| "reward_std": 0.4511892139911652, |
| "rewards/reward_accuracy/mean": 0.30390625, |
| "rewards/reward_accuracy/std": 0.4511892139911652, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7031883358955384, |
| "sampling/importance_sampling_ratio/mean": 0.9967585980892182, |
| "sampling/importance_sampling_ratio/min": 0.5146667748689652, |
| "sampling/sampling_logp_difference/max": 0.6616849541664124, |
| "sampling/sampling_logp_difference/mean": 0.0013093462097458542, |
| "step": 1840, |
| "step_time": 4.972000019508414 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 84.7, |
| "completions/max_terminated_length": 84.7, |
| "completions/mean_length": 50.409375, |
| "completions/mean_terminated_length": 50.409375, |
| "completions/min_length": 28.4, |
| "completions/min_terminated_length": 28.4, |
| "entropy": 0.0059205536417721305, |
| "epoch": 4.780361757105943, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8151e-06, |
| "loss": -0.0004, |
| "num_tokens": 101832471.0, |
| "reward": 0.3609375223517418, |
| "reward_std": 0.43351986110210416, |
| "rewards/reward_accuracy/mean": 0.2609375, |
| "rewards/reward_accuracy/std": 0.43351986110210416, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5579183220863342, |
| "sampling/importance_sampling_ratio/mean": 0.9899882018566132, |
| "sampling/importance_sampling_ratio/min": 0.43351308107376096, |
| "sampling/sampling_logp_difference/max": 0.9042840480804444, |
| "sampling/sampling_logp_difference/mean": 0.001239886405528523, |
| "step": 1850, |
| "step_time": 5.159000160568394 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 52.0484375, |
| "completions/mean_terminated_length": 52.0484375, |
| "completions/min_length": 27.4, |
| "completions/min_terminated_length": 27.4, |
| "entropy": 0.004938754494196474, |
| "epoch": 4.8062015503875966, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.814100000000001e-06, |
| "loss": 0.0009, |
| "num_tokens": 102386901.0, |
| "reward": 0.4015625223517418, |
| "reward_std": 0.44671647250652313, |
| "rewards/reward_accuracy/mean": 0.3015625, |
| "rewards/reward_accuracy/std": 0.4467164695262909, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5597286581993104, |
| "sampling/importance_sampling_ratio/mean": 0.9891061007976532, |
| "sampling/importance_sampling_ratio/min": 0.3607055801898241, |
| "sampling/sampling_logp_difference/max": 1.304043710231781, |
| "sampling/sampling_logp_difference/mean": 0.0015351104753790422, |
| "step": 1860, |
| "step_time": 4.875464421254582 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.4, |
| "completions/max_terminated_length": 72.4, |
| "completions/mean_length": 49.66171875, |
| "completions/mean_terminated_length": 49.66171875, |
| "completions/min_length": 28.1, |
| "completions/min_terminated_length": 28.1, |
| "entropy": 0.004086325139724067, |
| "epoch": 4.832041343669251, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.78515625, |
| "learning_rate": 9.813100000000001e-06, |
| "loss": 0.0004, |
| "num_tokens": 102935316.0, |
| "reward": 0.4140625223517418, |
| "reward_std": 0.45125689208507536, |
| "rewards/reward_accuracy/mean": 0.3140625, |
| "rewards/reward_accuracy/std": 0.45125689208507536, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.590861690044403, |
| "sampling/importance_sampling_ratio/mean": 0.9921106696128845, |
| "sampling/importance_sampling_ratio/min": 0.5777311131358147, |
| "sampling/sampling_logp_difference/max": 0.6858186662197113, |
| "sampling/sampling_logp_difference/mean": 0.0007673815125599504, |
| "step": 1870, |
| "step_time": 4.954836212750524 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 95.3, |
| "completions/max_terminated_length": 95.3, |
| "completions/mean_length": 49.99921875, |
| "completions/mean_terminated_length": 49.99921875, |
| "completions/min_length": 25.6, |
| "completions/min_terminated_length": 25.6, |
| "entropy": 0.0046964699718955675, |
| "epoch": 4.857881136950905, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.54296875, |
| "learning_rate": 9.8121e-06, |
| "loss": 0.0004, |
| "num_tokens": 103483835.0, |
| "reward": 0.4085937738418579, |
| "reward_std": 0.44743515849113463, |
| "rewards/reward_accuracy/mean": 0.30859375, |
| "rewards/reward_accuracy/std": 0.4474351555109024, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.832438886165619, |
| "sampling/importance_sampling_ratio/mean": 1.0013446033000946, |
| "sampling/importance_sampling_ratio/min": 0.5677409827709198, |
| "sampling/sampling_logp_difference/max": 0.7018273770809174, |
| "sampling/sampling_logp_difference/mean": 0.0009846643137279899, |
| "step": 1880, |
| "step_time": 5.244565638597123 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.7, |
| "completions/max_terminated_length": 71.7, |
| "completions/mean_length": 47.81796875, |
| "completions/mean_terminated_length": 47.81796875, |
| "completions/min_length": 27.6, |
| "completions/min_terminated_length": 27.6, |
| "entropy": 0.0039438518045244566, |
| "epoch": 4.883720930232558, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 1.6015625, |
| "learning_rate": 9.811100000000002e-06, |
| "loss": 0.0002, |
| "num_tokens": 104023746.0, |
| "reward": 0.4171875223517418, |
| "reward_std": 0.44434729516506194, |
| "rewards/reward_accuracy/mean": 0.3171875, |
| "rewards/reward_accuracy/std": 0.44434729516506194, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3295316100120544, |
| "sampling/importance_sampling_ratio/mean": 0.9979717075824738, |
| "sampling/importance_sampling_ratio/min": 0.48423912674188613, |
| "sampling/sampling_logp_difference/max": 0.7400215029716491, |
| "sampling/sampling_logp_difference/mean": 0.0008656961115775629, |
| "step": 1890, |
| "step_time": 4.915977371204645 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.7, |
| "completions/max_terminated_length": 70.7, |
| "completions/mean_length": 50.31328125, |
| "completions/mean_terminated_length": 50.31328125, |
| "completions/min_length": 29.0, |
| "completions/min_terminated_length": 29.0, |
| "entropy": 0.003266777929093223, |
| "epoch": 4.909560723514212, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.810100000000001e-06, |
| "loss": 0.0022, |
| "num_tokens": 104573875.0, |
| "reward": 0.4281250238418579, |
| "reward_std": 0.4557654678821564, |
| "rewards/reward_accuracy/mean": 0.328125, |
| "rewards/reward_accuracy/std": 0.4557654678821564, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5205909490585328, |
| "sampling/importance_sampling_ratio/mean": 0.9887771904468536, |
| "sampling/importance_sampling_ratio/min": 0.38444956541061404, |
| "sampling/sampling_logp_difference/max": 0.8944741606712341, |
| "sampling/sampling_logp_difference/mean": 0.001058134448248893, |
| "step": 1900, |
| "step_time": 4.872312090266496 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.3, |
| "completions/max_terminated_length": 72.3, |
| "completions/mean_length": 49.8078125, |
| "completions/mean_terminated_length": 49.8078125, |
| "completions/min_length": 26.8, |
| "completions/min_terminated_length": 26.8, |
| "entropy": 0.003920476123494154, |
| "epoch": 4.935400516795866, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 1.1875, |
| "learning_rate": 9.8091e-06, |
| "loss": 0.0056, |
| "num_tokens": 105121213.0, |
| "reward": 0.3953125238418579, |
| "reward_std": 0.4362353295087814, |
| "rewards/reward_accuracy/mean": 0.2953125, |
| "rewards/reward_accuracy/std": 0.4362353265285492, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6535009741783142, |
| "sampling/importance_sampling_ratio/mean": 0.9959949851036072, |
| "sampling/importance_sampling_ratio/min": 0.5980521768331528, |
| "sampling/sampling_logp_difference/max": 0.578728187084198, |
| "sampling/sampling_logp_difference/mean": 0.0009166882955469191, |
| "step": 1910, |
| "step_time": 4.850527582550422 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.4, |
| "completions/max_terminated_length": 73.4, |
| "completions/mean_length": 51.44609375, |
| "completions/mean_terminated_length": 51.44609375, |
| "completions/min_length": 28.5, |
| "completions/min_terminated_length": 28.5, |
| "entropy": 0.0031861552606642363, |
| "epoch": 4.961240310077519, |
| "frac_reward_zero_std": 0.95625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8081e-06, |
| "loss": -0.0018, |
| "num_tokens": 105674592.0, |
| "reward": 0.4242187738418579, |
| "reward_std": 0.4479935348033905, |
| "rewards/reward_accuracy/mean": 0.32421875, |
| "rewards/reward_accuracy/std": 0.4479935348033905, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.472868835926056, |
| "sampling/importance_sampling_ratio/mean": 0.9974490702152252, |
| "sampling/importance_sampling_ratio/min": 0.5453327804803848, |
| "sampling/sampling_logp_difference/max": 0.6962130308151245, |
| "sampling/sampling_logp_difference/mean": 0.0008355898258741945, |
| "step": 1920, |
| "step_time": 5.030009154253639 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 50.4421875, |
| "completions/mean_terminated_length": 50.4421875, |
| "completions/min_length": 25.9, |
| "completions/min_terminated_length": 25.9, |
| "entropy": 0.003018746853376797, |
| "epoch": 4.987080103359173, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.94140625, |
| "learning_rate": 9.8071e-06, |
| "loss": 0.0022, |
| "num_tokens": 106225502.0, |
| "reward": 0.3437500223517418, |
| "reward_std": 0.42159993946552277, |
| "rewards/reward_accuracy/mean": 0.24375, |
| "rewards/reward_accuracy/std": 0.42159993946552277, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5425675630569458, |
| "sampling/importance_sampling_ratio/mean": 0.9989052414894104, |
| "sampling/importance_sampling_ratio/min": 0.6356425821781159, |
| "sampling/sampling_logp_difference/max": 0.6597748100757599, |
| "sampling/sampling_logp_difference/mean": 0.0008394642500206828, |
| "step": 1930, |
| "step_time": 4.960596339940094 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 82.1, |
| "completions/max_terminated_length": 82.1, |
| "completions/mean_length": 50.38046875, |
| "completions/mean_terminated_length": 50.38046875, |
| "completions/min_length": 25.9, |
| "completions/min_terminated_length": 25.9, |
| "entropy": 0.0037288604242348812, |
| "epoch": 5.012919896640827, |
| "frac_reward_zero_std": 0.94375, |
| "grad_norm": 1.5390625, |
| "learning_rate": 9.806100000000001e-06, |
| "loss": -0.0036, |
| "num_tokens": 106773149.0, |
| "reward": 0.3554687723517418, |
| "reward_std": 0.4271271854639053, |
| "rewards/reward_accuracy/mean": 0.25546875, |
| "rewards/reward_accuracy/std": 0.4271271854639053, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5781564593315125, |
| "sampling/importance_sampling_ratio/mean": 1.0008474111557006, |
| "sampling/importance_sampling_ratio/min": 0.6148100361227989, |
| "sampling/sampling_logp_difference/max": 0.7247543394565582, |
| "sampling/sampling_logp_difference/mean": 0.00084852792606398, |
| "step": 1940, |
| "step_time": 5.031182863446884 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.7, |
| "completions/max_terminated_length": 71.7, |
| "completions/mean_length": 49.665625, |
| "completions/mean_terminated_length": 49.665625, |
| "completions/min_length": 29.3, |
| "completions/min_terminated_length": 29.3, |
| "entropy": 0.0028814565151606074, |
| "epoch": 5.038759689922481, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8051e-06, |
| "loss": 0.0097, |
| "num_tokens": 107320657.0, |
| "reward": 0.3828125223517418, |
| "reward_std": 0.43953052163124084, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.43953052163124084, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.673397958278656, |
| "sampling/importance_sampling_ratio/mean": 0.9986977219581604, |
| "sampling/importance_sampling_ratio/min": 0.6449146032333374, |
| "sampling/sampling_logp_difference/max": 0.7142746623605489, |
| "sampling/sampling_logp_difference/mean": 0.0009253416098545131, |
| "step": 1950, |
| "step_time": 5.020455218292772 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.7, |
| "completions/max_terminated_length": 72.7, |
| "completions/mean_length": 50.6765625, |
| "completions/mean_terminated_length": 50.6765625, |
| "completions/min_length": 29.2, |
| "completions/min_terminated_length": 29.2, |
| "entropy": 0.002505797770436402, |
| "epoch": 5.064599483204135, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 2.078125, |
| "learning_rate": 9.804100000000002e-06, |
| "loss": 0.0077, |
| "num_tokens": 107872251.0, |
| "reward": 0.4578125238418579, |
| "reward_std": 0.46555028259754183, |
| "rewards/reward_accuracy/mean": 0.3578125, |
| "rewards/reward_accuracy/std": 0.46555028259754183, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.604396688938141, |
| "sampling/importance_sampling_ratio/mean": 0.9874807596206665, |
| "sampling/importance_sampling_ratio/min": 0.4209839262068272, |
| "sampling/sampling_logp_difference/max": 1.1654183328151704, |
| "sampling/sampling_logp_difference/mean": 0.0012279571063118055, |
| "step": 1960, |
| "step_time": 4.93271592867095 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.7, |
| "completions/max_terminated_length": 73.7, |
| "completions/mean_length": 51.821875, |
| "completions/mean_terminated_length": 51.821875, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.0024579999014349594, |
| "epoch": 5.090439276485788, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.803100000000001e-06, |
| "loss": -0.0003, |
| "num_tokens": 108427151.0, |
| "reward": 0.3757812723517418, |
| "reward_std": 0.4283652424812317, |
| "rewards/reward_accuracy/mean": 0.27578125, |
| "rewards/reward_accuracy/std": 0.4283652424812317, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4334608435630798, |
| "sampling/importance_sampling_ratio/mean": 1.0000827014446259, |
| "sampling/importance_sampling_ratio/min": 0.6131557986140251, |
| "sampling/sampling_logp_difference/max": 0.7788373801857233, |
| "sampling/sampling_logp_difference/mean": 0.0007768766743538435, |
| "step": 1970, |
| "step_time": 5.030842458806001 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.2, |
| "completions/max_terminated_length": 73.2, |
| "completions/mean_length": 50.0015625, |
| "completions/mean_terminated_length": 50.0015625, |
| "completions/min_length": 25.0, |
| "completions/min_terminated_length": 25.0, |
| "entropy": 0.0026729062007689207, |
| "epoch": 5.116279069767442, |
| "frac_reward_zero_std": 0.96875, |
| "grad_norm": 1.9375, |
| "learning_rate": 9.8021e-06, |
| "loss": 0.0053, |
| "num_tokens": 108975153.0, |
| "reward": 0.4054687723517418, |
| "reward_std": 0.4535236954689026, |
| "rewards/reward_accuracy/mean": 0.30546875, |
| "rewards/reward_accuracy/std": 0.4535236954689026, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4828598260879517, |
| "sampling/importance_sampling_ratio/mean": 0.9920314073562622, |
| "sampling/importance_sampling_ratio/min": 0.5722768306732178, |
| "sampling/sampling_logp_difference/max": 0.6903791543096304, |
| "sampling/sampling_logp_difference/mean": 0.000976664980771602, |
| "step": 1980, |
| "step_time": 4.9850892358692365 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 73.6, |
| "completions/max_terminated_length": 73.6, |
| "completions/mean_length": 51.9046875, |
| "completions/mean_terminated_length": 51.9046875, |
| "completions/min_length": 26.4, |
| "completions/min_terminated_length": 26.4, |
| "entropy": 0.0027826305195048917, |
| "epoch": 5.142118863049095, |
| "frac_reward_zero_std": 0.9625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.801100000000002e-06, |
| "loss": 0.007, |
| "num_tokens": 109530239.0, |
| "reward": 0.3992187738418579, |
| "reward_std": 0.4490850269794464, |
| "rewards/reward_accuracy/mean": 0.29921875, |
| "rewards/reward_accuracy/std": 0.4490850269794464, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4551566004753114, |
| "sampling/importance_sampling_ratio/mean": 0.9906647861003876, |
| "sampling/importance_sampling_ratio/min": 0.51825051009655, |
| "sampling/sampling_logp_difference/max": 0.8633471459150315, |
| "sampling/sampling_logp_difference/mean": 0.0009434973864699714, |
| "step": 1990, |
| "step_time": 5.077982176025398 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 72.9, |
| "completions/max_terminated_length": 72.9, |
| "completions/mean_length": 50.0078125, |
| "completions/mean_terminated_length": 50.0078125, |
| "completions/min_length": 26.7, |
| "completions/min_terminated_length": 26.7, |
| "entropy": 0.0024873181469502017, |
| "epoch": 5.167958656330749, |
| "frac_reward_zero_std": 0.98125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.800100000000001e-06, |
| "loss": -0.0072, |
| "num_tokens": 110078233.0, |
| "reward": 0.3742187723517418, |
| "reward_std": 0.43424488604068756, |
| "rewards/reward_accuracy/mean": 0.27421875, |
| "rewards/reward_accuracy/std": 0.43424488604068756, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4561767101287841, |
| "sampling/importance_sampling_ratio/mean": 0.9994723081588746, |
| "sampling/importance_sampling_ratio/min": 0.6241671934723854, |
| "sampling/sampling_logp_difference/max": 0.637858933955431, |
| "sampling/sampling_logp_difference/mean": 0.0007665604098292533, |
| "step": 2000, |
| "step_time": 4.9724137298297135 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 100000, |
| "num_input_tokens_seen": 110078233, |
| "num_train_epochs": 259, |
| "save_steps": 100, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|