| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5434782608695652, |
| "eval_steps": 500, |
| "global_step": 200, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2017.0, |
| "completions/mean_length": 1671.2799072265625, |
| "completions/mean_terminated_length": 1398.4827880859375, |
| "completions/min_length": 992.0, |
| "completions/min_terminated_length": 992.0, |
| "entropy": 0.21248915791511536, |
| "epoch": 0.002717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016046486794948578, |
| "learning_rate": 0.0, |
| "loss": 0.08257194608449936, |
| "num_tokens": 86304.0, |
| "reward": -0.21605467796325684, |
| "reward_std": 0.656523585319519, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.8160547018051147, |
| "rewards/length_penalty/std": 0.18964654207229614, |
| "sampling/importance_sampling_ratio/max": 1.5222556591033936, |
| "sampling/importance_sampling_ratio/mean": 0.9927070140838623, |
| "sampling/importance_sampling_ratio/min": 0.6435883045196533, |
| "sampling/sampling_logp_difference/max": 0.44069600105285645, |
| "sampling/sampling_logp_difference/mean": 0.013988969847559929, |
| "step": 1, |
| "step_time": 27.354138373862952 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.6800000071525574, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1991.0, |
| "completions/mean_length": 1941.47998046875, |
| "completions/mean_terminated_length": 1715.125, |
| "completions/min_length": 1079.0, |
| "completions/min_terminated_length": 1079.0, |
| "entropy": 0.2905632793903351, |
| "epoch": 0.005434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017462793737649918, |
| "learning_rate": 5e-06, |
| "loss": 0.07610104233026505, |
| "num_tokens": 186618.0, |
| "reward": -0.7279882431030273, |
| "reward_std": 0.4917677640914917, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.9479882717132568, |
| "rewards/length_penalty/std": 0.10518474876880646, |
| "sampling/importance_sampling_ratio/max": 1.450933814048767, |
| "sampling/importance_sampling_ratio/mean": 0.9901063442230225, |
| "sampling/importance_sampling_ratio/min": 0.3688630163669586, |
| "sampling/sampling_logp_difference/max": 0.9973299503326416, |
| "sampling/sampling_logp_difference/mean": 0.01780563034117222, |
| "step": 2, |
| "step_time": 28.93621321511455 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008385480789002032, |
| "clip_ratio/high_mean": 0.0008385480789002032, |
| "clip_ratio/low_mean": 0.00012290228332858532, |
| "clip_ratio/low_min": 0.00012290228332858532, |
| "clip_ratio/region_mean": 0.0009614503418561071, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2047.0, |
| "completions/mean_length": 1629.93994140625, |
| "completions/mean_terminated_length": 1512.025634765625, |
| "completions/min_length": 809.0, |
| "completions/min_terminated_length": 809.0, |
| "entropy": 0.2568131685256958, |
| "epoch": 0.008152173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017640408128499985, |
| "learning_rate": 1e-05, |
| "loss": 0.14312560856342316, |
| "num_tokens": 270385.0, |
| "reward": -0.015869140625, |
| "reward_std": 0.5487767457962036, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.7958691120147705, |
| "rewards/length_penalty/std": 0.18548113107681274, |
| "sampling/importance_sampling_ratio/max": 1.5610862970352173, |
| "sampling/importance_sampling_ratio/mean": 0.991174042224884, |
| "sampling/importance_sampling_ratio/min": 0.6428658366203308, |
| "sampling/sampling_logp_difference/max": 0.44538187980651855, |
| "sampling/sampling_logp_difference/mean": 0.016407834365963936, |
| "step": 3, |
| "step_time": 27.082281265873462 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005171060445718467, |
| "clip_ratio/high_mean": 0.0005171060445718467, |
| "clip_ratio/low_mean": 0.0001079634006600827, |
| "clip_ratio/low_min": 0.0001079634006600827, |
| "clip_ratio/region_mean": 0.0006250694510526955, |
| "completions/clipped_ratio": 0.25999999046325684, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1991.0, |
| "completions/mean_length": 1577.8199462890625, |
| "completions/mean_terminated_length": 1412.6217041015625, |
| "completions/min_length": 655.0, |
| "completions/min_terminated_length": 655.0, |
| "entropy": 0.2619533360004425, |
| "epoch": 0.010869565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017685169354081154, |
| "learning_rate": 1.5e-05, |
| "loss": 0.1186390221118927, |
| "num_tokens": 352176.0, |
| "reward": -0.03041992150247097, |
| "reward_std": 0.5997232794761658, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.7704198956489563, |
| "rewards/length_penalty/std": 0.20377306640148163, |
| "sampling/importance_sampling_ratio/max": 1.679875135421753, |
| "sampling/importance_sampling_ratio/mean": 0.9910148978233337, |
| "sampling/importance_sampling_ratio/min": 0.5549276471138, |
| "sampling/sampling_logp_difference/max": 0.5889174938201904, |
| "sampling/sampling_logp_difference/mean": 0.01687830127775669, |
| "step": 4, |
| "step_time": 27.36521916766651 |
| }, |
| { |
| "clip_ratio/high_max": 0.00031414994155056777, |
| "clip_ratio/high_mean": 0.00031414994155056777, |
| "clip_ratio/low_mean": 0.00013472853970597498, |
| "clip_ratio/low_min": 0.00013472853970597498, |
| "clip_ratio/region_mean": 0.0004488784761633724, |
| "completions/clipped_ratio": 0.7999999523162842, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1748.0, |
| "completions/mean_length": 1920.47998046875, |
| "completions/mean_terminated_length": 1410.4000244140625, |
| "completions/min_length": 1134.0, |
| "completions/min_terminated_length": 1134.0, |
| "entropy": 0.28744050562381745, |
| "epoch": 0.01358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017345719039440155, |
| "learning_rate": 2e-05, |
| "loss": 0.0791063904762268, |
| "num_tokens": 451910.0, |
| "reward": -0.7177343368530273, |
| "reward_std": 0.5403889417648315, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.9377343654632568, |
| "rewards/length_penalty/std": 0.13342204689979553, |
| "sampling/importance_sampling_ratio/max": 1.4594295024871826, |
| "sampling/importance_sampling_ratio/mean": 0.9900663495063782, |
| "sampling/importance_sampling_ratio/min": 0.6219714283943176, |
| "sampling/sampling_logp_difference/max": 0.47486114501953125, |
| "sampling/sampling_logp_difference/mean": 0.018159378319978714, |
| "step": 5, |
| "step_time": 28.97891273209825 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004946111119352281, |
| "clip_ratio/high_mean": 0.0004946111119352281, |
| "clip_ratio/low_mean": 0.00017250097589567305, |
| "clip_ratio/low_min": 0.00017250097589567305, |
| "clip_ratio/region_mean": 0.0006671120820101351, |
| "completions/clipped_ratio": 0.47999998927116394, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1947.0, |
| "completions/mean_length": 1633.93994140625, |
| "completions/mean_terminated_length": 1251.7308349609375, |
| "completions/min_length": 701.0, |
| "completions/min_terminated_length": 701.0, |
| "entropy": 0.23896483480930328, |
| "epoch": 0.016304347826086956, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017287233844399452, |
| "learning_rate": 2.5e-05, |
| "loss": 0.09624812752008438, |
| "num_tokens": 536597.0, |
| "reward": -0.2578222453594208, |
| "reward_std": 0.707152783870697, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.7978222370147705, |
| "rewards/length_penalty/std": 0.23831933736801147, |
| "sampling/importance_sampling_ratio/max": 1.5380786657333374, |
| "sampling/importance_sampling_ratio/mean": 0.9916631579399109, |
| "sampling/importance_sampling_ratio/min": 0.6339684128761292, |
| "sampling/sampling_logp_difference/max": 0.45575618743896484, |
| "sampling/sampling_logp_difference/mean": 0.015873005613684654, |
| "step": 6, |
| "step_time": 27.311647549970075 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007668758626095951, |
| "clip_ratio/high_mean": 0.0007668758626095951, |
| "clip_ratio/low_mean": 0.00020238555807736703, |
| "clip_ratio/low_min": 0.00020238555807736703, |
| "clip_ratio/region_mean": 0.0009692614316008985, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2024.0, |
| "completions/mean_length": 1703.1199951171875, |
| "completions/mean_terminated_length": 1525.45458984375, |
| "completions/min_length": 786.0, |
| "completions/min_terminated_length": 786.0, |
| "entropy": 0.2471608877182007, |
| "epoch": 0.019021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017884649336338043, |
| "learning_rate": 3e-05, |
| "loss": 0.10157303512096405, |
| "num_tokens": 623983.0, |
| "reward": -0.1716015636920929, |
| "reward_std": 0.6160923838615417, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.8316015601158142, |
| "rewards/length_penalty/std": 0.18369081616401672, |
| "sampling/importance_sampling_ratio/max": 1.7520934343338013, |
| "sampling/importance_sampling_ratio/mean": 0.9914238452911377, |
| "sampling/importance_sampling_ratio/min": 0.600068211555481, |
| "sampling/sampling_logp_difference/max": 0.5608112812042236, |
| "sampling/sampling_logp_difference/mean": 0.015808574855327606, |
| "step": 7, |
| "step_time": 27.56733594578691 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017992005741689353, |
| "clip_ratio/high_mean": 0.00017992005741689353, |
| "clip_ratio/low_mean": 2.3250406957231463e-05, |
| "clip_ratio/low_min": 2.3250406957231463e-05, |
| "clip_ratio/region_mean": 0.00020317046291893349, |
| "completions/clipped_ratio": 0.6200000047683716, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 1789.919921875, |
| "completions/mean_terminated_length": 1368.8421630859375, |
| "completions/min_length": 663.0, |
| "completions/min_terminated_length": 663.0, |
| "entropy": 0.2106441378593445, |
| "epoch": 0.021739130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016445742920041084, |
| "learning_rate": 3.5e-05, |
| "loss": 0.09301691502332687, |
| "num_tokens": 715629.0, |
| "reward": -0.4939843714237213, |
| "reward_std": 0.6623950600624084, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.8739843964576721, |
| "rewards/length_penalty/std": 0.1972581446170807, |
| "sampling/importance_sampling_ratio/max": 1.4558980464935303, |
| "sampling/importance_sampling_ratio/mean": 0.9929009675979614, |
| "sampling/importance_sampling_ratio/min": 0.6304246187210083, |
| "sampling/sampling_logp_difference/max": 0.4613616466522217, |
| "sampling/sampling_logp_difference/mean": 0.013887309469282627, |
| "step": 8, |
| "step_time": 28.2383045849856 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007567958615254611, |
| "clip_ratio/high_mean": 0.0007567958615254611, |
| "clip_ratio/low_mean": 0.000174625517684035, |
| "clip_ratio/low_min": 0.000174625517684035, |
| "clip_ratio/region_mean": 0.0009314213762991131, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1980.0, |
| "completions/mean_length": 1670.1199951171875, |
| "completions/mean_terminated_length": 1475.45458984375, |
| "completions/min_length": 750.0, |
| "completions/min_terminated_length": 750.0, |
| "entropy": 0.2428219199180603, |
| "epoch": 0.024456521739130436, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018087439239025116, |
| "learning_rate": 4e-05, |
| "loss": 0.13151915371418, |
| "num_tokens": 801605.0, |
| "reward": -0.29548826813697815, |
| "reward_std": 0.6281818151473999, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.8154882788658142, |
| "rewards/length_penalty/std": 0.18004027009010315, |
| "sampling/importance_sampling_ratio/max": 1.6334065198898315, |
| "sampling/importance_sampling_ratio/mean": 0.9916205406188965, |
| "sampling/importance_sampling_ratio/min": 0.6332058310508728, |
| "sampling/sampling_logp_difference/max": 0.4906677007675171, |
| "sampling/sampling_logp_difference/mean": 0.01592734456062317, |
| "step": 9, |
| "step_time": 27.450346793280914 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005264588573481888, |
| "clip_ratio/high_mean": 0.0005264588573481888, |
| "clip_ratio/low_mean": 0.0001535123068606481, |
| "clip_ratio/low_min": 0.0001535123068606481, |
| "clip_ratio/region_mean": 0.0006799711438361556, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1937.0, |
| "completions/mean_length": 1445.43994140625, |
| "completions/mean_terminated_length": 1275.4871826171875, |
| "completions/min_length": 621.0, |
| "completions/min_terminated_length": 621.0, |
| "entropy": 0.24449434280395507, |
| "epoch": 0.02717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017047857865691185, |
| "learning_rate": 4.5e-05, |
| "loss": 0.15399205684661865, |
| "num_tokens": 878697.0, |
| "reward": 0.014218749478459358, |
| "reward_std": 0.6368762254714966, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.7057812213897705, |
| "rewards/length_penalty/std": 0.23674166202545166, |
| "sampling/importance_sampling_ratio/max": 1.5615642070770264, |
| "sampling/importance_sampling_ratio/mean": 0.9914027452468872, |
| "sampling/importance_sampling_ratio/min": 0.5404947400093079, |
| "sampling/sampling_logp_difference/max": 0.6152703762054443, |
| "sampling/sampling_logp_difference/mean": 0.01580055244266987, |
| "step": 10, |
| "step_time": 27.534237888874486 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006183574412716552, |
| "clip_ratio/high_mean": 0.0006183574412716552, |
| "clip_ratio/low_mean": 0.00014809061540290713, |
| "clip_ratio/low_min": 0.00014809061540290713, |
| "clip_ratio/region_mean": 0.0007664480479434132, |
| "completions/clipped_ratio": 0.47999998927116394, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1947.0, |
| "completions/mean_length": 1872.9599609375, |
| "completions/mean_terminated_length": 1711.3846435546875, |
| "completions/min_length": 1080.0, |
| "completions/min_terminated_length": 1080.0, |
| "entropy": 0.2463698834180832, |
| "epoch": 0.029891304347826088, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017627008259296417, |
| "learning_rate": 5e-05, |
| "loss": 0.0819278135895729, |
| "num_tokens": 974765.0, |
| "reward": -0.39453125, |
| "reward_std": 0.5928694605827332, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.9145312309265137, |
| "rewards/length_penalty/std": 0.11434958875179291, |
| "sampling/importance_sampling_ratio/max": 1.7603743076324463, |
| "sampling/importance_sampling_ratio/mean": 0.9915382266044617, |
| "sampling/importance_sampling_ratio/min": 0.6209133863449097, |
| "sampling/sampling_logp_difference/max": 0.5655264854431152, |
| "sampling/sampling_logp_difference/mean": 0.015730150043964386, |
| "step": 11, |
| "step_time": 28.32604410406202 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005333822926331777, |
| "clip_ratio/high_mean": 0.0005333822926331777, |
| "clip_ratio/low_mean": 0.000139446763205342, |
| "clip_ratio/low_min": 0.000139446763205342, |
| "clip_ratio/region_mean": 0.0006728290492901579, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2017.0, |
| "completions/mean_length": 1717.8800048828125, |
| "completions/mean_terminated_length": 1478.82763671875, |
| "completions/min_length": 845.0, |
| "completions/min_terminated_length": 845.0, |
| "entropy": 0.28277077674865725, |
| "epoch": 0.03260869565217391, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018042029812932014, |
| "learning_rate": 5e-05, |
| "loss": 0.132859006524086, |
| "num_tokens": 1063299.0, |
| "reward": -0.23880858719348907, |
| "reward_std": 0.6367524862289429, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.8388085961341858, |
| "rewards/length_penalty/std": 0.17020854353904724, |
| "sampling/importance_sampling_ratio/max": 1.6479967832565308, |
| "sampling/importance_sampling_ratio/mean": 0.9900858402252197, |
| "sampling/importance_sampling_ratio/min": 0.5561046004295349, |
| "sampling/sampling_logp_difference/max": 0.586798906326294, |
| "sampling/sampling_logp_difference/mean": 0.01785198599100113, |
| "step": 12, |
| "step_time": 28.157431120984256 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006276467058341951, |
| "clip_ratio/high_mean": 0.0006276467058341951, |
| "clip_ratio/low_mean": 0.00014667867799289525, |
| "clip_ratio/low_min": 0.00014667867799289525, |
| "clip_ratio/region_mean": 0.0007743253721855581, |
| "completions/clipped_ratio": 0.3199999928474426, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1965.0, |
| "completions/mean_length": 1650.43994140625, |
| "completions/mean_terminated_length": 1463.3529052734375, |
| "completions/min_length": 922.0, |
| "completions/min_terminated_length": 922.0, |
| "entropy": 0.2581978112459183, |
| "epoch": 0.035326086956521736, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018234722316265106, |
| "learning_rate": 5e-05, |
| "loss": 0.10529822111129761, |
| "num_tokens": 1148841.0, |
| "reward": -0.28587889671325684, |
| "reward_std": 0.6085616946220398, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.8058788776397705, |
| "rewards/length_penalty/std": 0.1748262494802475, |
| "sampling/importance_sampling_ratio/max": 1.6126469373703003, |
| "sampling/importance_sampling_ratio/mean": 0.9910162687301636, |
| "sampling/importance_sampling_ratio/min": 0.4064101576805115, |
| "sampling/sampling_logp_difference/max": 0.9003924131393433, |
| "sampling/sampling_logp_difference/mean": 0.016805455088615417, |
| "step": 13, |
| "step_time": 27.38471025787294 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007995524443686009, |
| "clip_ratio/high_mean": 0.0007995524443686009, |
| "clip_ratio/low_mean": 0.00011164293246110902, |
| "clip_ratio/low_min": 0.00011164293246110902, |
| "clip_ratio/region_mean": 0.0009111953899264335, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2044.0, |
| "completions/mean_length": 1757.1199951171875, |
| "completions/mean_terminated_length": 1466.239990234375, |
| "completions/min_length": 906.0, |
| "completions/min_terminated_length": 906.0, |
| "entropy": 0.2551351934671402, |
| "epoch": 0.03804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017962472513318062, |
| "learning_rate": 5e-05, |
| "loss": 0.11896330118179321, |
| "num_tokens": 1239257.0, |
| "reward": -0.3179687559604645, |
| "reward_std": 0.6411882638931274, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.8579687476158142, |
| "rewards/length_penalty/std": 0.18477807939052582, |
| "sampling/importance_sampling_ratio/max": 1.4596854448318481, |
| "sampling/importance_sampling_ratio/mean": 0.9910740852355957, |
| "sampling/importance_sampling_ratio/min": 0.5824056267738342, |
| "sampling/sampling_logp_difference/max": 0.5405881404876709, |
| "sampling/sampling_logp_difference/mean": 0.016177302226424217, |
| "step": 14, |
| "step_time": 28.40847225417383 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010078657185658812, |
| "clip_ratio/high_mean": 0.0010078657185658812, |
| "clip_ratio/low_mean": 9.516176214674488e-05, |
| "clip_ratio/low_min": 9.516176214674488e-05, |
| "clip_ratio/region_mean": 0.0011030274792574346, |
| "completions/clipped_ratio": 0.3199999928474426, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2015.0, |
| "completions/mean_length": 1612.699951171875, |
| "completions/mean_terminated_length": 1407.8529052734375, |
| "completions/min_length": 833.0, |
| "completions/min_terminated_length": 833.0, |
| "entropy": 0.19166844189167023, |
| "epoch": 0.04076086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018429262563586235, |
| "learning_rate": 5e-05, |
| "loss": 0.13294844329357147, |
| "num_tokens": 1322472.0, |
| "reward": -0.2874511778354645, |
| "reward_std": 0.6260359287261963, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.7874511480331421, |
| "rewards/length_penalty/std": 0.18555159866809845, |
| "sampling/importance_sampling_ratio/max": 1.6358195543289185, |
| "sampling/importance_sampling_ratio/mean": 0.9931573271751404, |
| "sampling/importance_sampling_ratio/min": 0.5374121069908142, |
| "sampling/sampling_logp_difference/max": 0.6209900379180908, |
| "sampling/sampling_logp_difference/mean": 0.012716456316411495, |
| "step": 15, |
| "step_time": 27.32731635682285 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007239608676172793, |
| "clip_ratio/high_mean": 0.0007239608676172793, |
| "clip_ratio/low_mean": 7.827363151591271e-05, |
| "clip_ratio/low_min": 7.827363151591271e-05, |
| "clip_ratio/region_mean": 0.0008022345020435751, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2005.0, |
| "completions/mean_length": 1510.1400146484375, |
| "completions/mean_terminated_length": 1279.6285400390625, |
| "completions/min_length": 580.0, |
| "completions/min_terminated_length": 580.0, |
| "entropy": 0.2445146143436432, |
| "epoch": 0.043478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017646517604589462, |
| "learning_rate": 5e-05, |
| "loss": 0.11172395944595337, |
| "num_tokens": 1402079.0, |
| "reward": -0.037373047322034836, |
| "reward_std": 0.658214807510376, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.7373730540275574, |
| "rewards/length_penalty/std": 0.24118255078792572, |
| "sampling/importance_sampling_ratio/max": 1.5615642070770264, |
| "sampling/importance_sampling_ratio/mean": 0.9913490414619446, |
| "sampling/importance_sampling_ratio/min": 0.4396955668926239, |
| "sampling/sampling_logp_difference/max": 0.8216726779937744, |
| "sampling/sampling_logp_difference/mean": 0.015879882499575615, |
| "step": 16, |
| "step_time": 27.297420918243006 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012540776166133582, |
| "clip_ratio/high_mean": 0.0012540776166133582, |
| "clip_ratio/low_mean": 8.785259196884e-05, |
| "clip_ratio/low_min": 8.785259196884e-05, |
| "clip_ratio/region_mean": 0.0013419301714748145, |
| "completions/clipped_ratio": 0.5399999618530273, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2005.0, |
| "completions/mean_length": 1760.260009765625, |
| "completions/mean_terminated_length": 1422.478271484375, |
| "completions/min_length": 847.0, |
| "completions/min_terminated_length": 847.0, |
| "entropy": 0.23043523728847504, |
| "epoch": 0.04619565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0170022863894701, |
| "learning_rate": 5e-05, |
| "loss": 0.08355365693569183, |
| "num_tokens": 1492742.0, |
| "reward": -0.3995019495487213, |
| "reward_std": 0.668391227722168, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8595019578933716, |
| "rewards/length_penalty/std": 0.196068674325943, |
| "sampling/importance_sampling_ratio/max": 1.4671045541763306, |
| "sampling/importance_sampling_ratio/mean": 0.9916527271270752, |
| "sampling/importance_sampling_ratio/min": 0.5674610137939453, |
| "sampling/sampling_logp_difference/max": 0.5665832757949829, |
| "sampling/sampling_logp_difference/mean": 0.015275287441909313, |
| "step": 17, |
| "step_time": 27.841468269005418 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005929864069912583, |
| "clip_ratio/high_mean": 0.0005929864069912583, |
| "clip_ratio/low_mean": 0.0001409070915542543, |
| "clip_ratio/low_min": 0.0001409070915542543, |
| "clip_ratio/region_mean": 0.0007338934927247464, |
| "completions/clipped_ratio": 0.3799999952316284, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 1699.5599365234375, |
| "completions/mean_terminated_length": 1486.0, |
| "completions/min_length": 968.0, |
| "completions/min_terminated_length": 968.0, |
| "entropy": 0.2047373831272125, |
| "epoch": 0.04891304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017544275149703026, |
| "learning_rate": 5e-05, |
| "loss": 0.08399701118469238, |
| "num_tokens": 1580220.0, |
| "reward": -0.2698632776737213, |
| "reward_std": 0.6521944403648376, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.8298633098602295, |
| "rewards/length_penalty/std": 0.18863293528556824, |
| "sampling/importance_sampling_ratio/max": 1.4644768238067627, |
| "sampling/importance_sampling_ratio/mean": 0.9928147196769714, |
| "sampling/importance_sampling_ratio/min": 0.3731330931186676, |
| "sampling/sampling_logp_difference/max": 0.9858200550079346, |
| "sampling/sampling_logp_difference/mean": 0.01354706846177578, |
| "step": 18, |
| "step_time": 28.045440710848197 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004978766839485616, |
| "clip_ratio/high_mean": 0.0004978766839485616, |
| "clip_ratio/low_mean": 0.00010482178186066449, |
| "clip_ratio/low_min": 0.00010482178186066449, |
| "clip_ratio/region_mean": 0.0006026984483469278, |
| "completions/clipped_ratio": 0.4599999785423279, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2033.0, |
| "completions/mean_length": 1746.919921875, |
| "completions/mean_terminated_length": 1490.4444580078125, |
| "completions/min_length": 742.0, |
| "completions/min_terminated_length": 742.0, |
| "entropy": 0.2838850259780884, |
| "epoch": 0.051630434782608696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01966138929128647, |
| "learning_rate": 5e-05, |
| "loss": 0.10605239868164062, |
| "num_tokens": 1670996.0, |
| "reward": -0.31298828125, |
| "reward_std": 0.6574001312255859, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.8529883027076721, |
| "rewards/length_penalty/std": 0.2017289698123932, |
| "sampling/importance_sampling_ratio/max": 1.5019716024398804, |
| "sampling/importance_sampling_ratio/mean": 0.989974319934845, |
| "sampling/importance_sampling_ratio/min": 0.5443934202194214, |
| "sampling/sampling_logp_difference/max": 0.6080830693244934, |
| "sampling/sampling_logp_difference/mean": 0.018089720979332924, |
| "step": 19, |
| "step_time": 27.931005945894867 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004667813540436327, |
| "clip_ratio/high_mean": 0.0004667813540436327, |
| "clip_ratio/low_mean": 0.00013519662170438095, |
| "clip_ratio/low_min": 0.00013519662170438095, |
| "clip_ratio/region_mean": 0.0006019779946655035, |
| "completions/clipped_ratio": 0.35999998450279236, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2044.0, |
| "completions/mean_length": 1763.239990234375, |
| "completions/mean_terminated_length": 1603.0625, |
| "completions/min_length": 682.0, |
| "completions/min_terminated_length": 682.0, |
| "entropy": 0.2596771240234375, |
| "epoch": 0.05434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018117377534508705, |
| "learning_rate": 5e-05, |
| "loss": 0.10192456096410751, |
| "num_tokens": 1761748.0, |
| "reward": -0.24095702171325684, |
| "reward_std": 0.6079331636428833, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.8609570264816284, |
| "rewards/length_penalty/std": 0.16505935788154602, |
| "sampling/importance_sampling_ratio/max": 1.466619849205017, |
| "sampling/importance_sampling_ratio/mean": 0.9911731481552124, |
| "sampling/importance_sampling_ratio/min": 0.4321824312210083, |
| "sampling/sampling_logp_difference/max": 0.8389074802398682, |
| "sampling/sampling_logp_difference/mean": 0.016629504039883614, |
| "step": 20, |
| "step_time": 28.29574998607859 |
| }, |
| { |
| "clip_ratio/high_max": 0.000582252535969019, |
| "clip_ratio/high_mean": 0.000582252535969019, |
| "clip_ratio/low_mean": 0.0001413671998307109, |
| "clip_ratio/low_min": 0.0001413671998307109, |
| "clip_ratio/region_mean": 0.0007236197357997298, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1974.0, |
| "completions/mean_length": 1719.7799072265625, |
| "completions/mean_terminated_length": 1550.697021484375, |
| "completions/min_length": 870.0, |
| "completions/min_terminated_length": 870.0, |
| "entropy": 0.2195265144109726, |
| "epoch": 0.057065217391304345, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018685726448893547, |
| "learning_rate": 5e-05, |
| "loss": 0.0958474725484848, |
| "num_tokens": 1851147.0, |
| "reward": -0.17973633110523224, |
| "reward_std": 0.6063199043273926, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.8397363424301147, |
| "rewards/length_penalty/std": 0.16565275192260742, |
| "sampling/importance_sampling_ratio/max": 1.5404351949691772, |
| "sampling/importance_sampling_ratio/mean": 0.9925079345703125, |
| "sampling/importance_sampling_ratio/min": 0.6400713324546814, |
| "sampling/sampling_logp_difference/max": 0.4461756944656372, |
| "sampling/sampling_logp_difference/mean": 0.014328239485621452, |
| "step": 21, |
| "step_time": 27.761335090035573 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006343422865029424, |
| "clip_ratio/high_mean": 0.0006343422865029424, |
| "clip_ratio/low_mean": 0.00018607097445055842, |
| "clip_ratio/low_min": 0.00018607097445055842, |
| "clip_ratio/region_mean": 0.0008204132318496704, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1915.0, |
| "completions/mean_length": 1489.219970703125, |
| "completions/mean_terminated_length": 1249.742919921875, |
| "completions/min_length": 755.0, |
| "completions/min_terminated_length": 755.0, |
| "entropy": 0.25570841133594513, |
| "epoch": 0.059782608695652176, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020179560407996178, |
| "learning_rate": 5e-05, |
| "loss": 0.10957732051610947, |
| "num_tokens": 1930678.0, |
| "reward": 0.01284179650247097, |
| "reward_std": 0.6249299645423889, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.7271581888198853, |
| "rewards/length_penalty/std": 0.22234591841697693, |
| "sampling/importance_sampling_ratio/max": 1.452115774154663, |
| "sampling/importance_sampling_ratio/mean": 0.9913490414619446, |
| "sampling/importance_sampling_ratio/min": 0.569891631603241, |
| "sampling/sampling_logp_difference/max": 0.5623090267181396, |
| "sampling/sampling_logp_difference/mean": 0.016323108226060867, |
| "step": 22, |
| "step_time": 28.038484191987664 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008312713471241295, |
| "clip_ratio/high_mean": 0.0008312713471241295, |
| "clip_ratio/low_mean": 0.00018834287475328894, |
| "clip_ratio/low_min": 0.00018834287475328894, |
| "clip_ratio/region_mean": 0.0010196142247878015, |
| "completions/clipped_ratio": 0.35999998450279236, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2042.0, |
| "completions/mean_length": 1594.8800048828125, |
| "completions/mean_terminated_length": 1340.0, |
| "completions/min_length": 802.0, |
| "completions/min_terminated_length": 802.0, |
| "entropy": 0.27381070256233214, |
| "epoch": 0.0625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018540026620030403, |
| "learning_rate": 5e-05, |
| "loss": 0.11850632727146149, |
| "num_tokens": 2016542.0, |
| "reward": -0.2787500023841858, |
| "reward_std": 0.6703850626945496, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.7787500023841858, |
| "rewards/length_penalty/std": 0.2127879559993744, |
| "sampling/importance_sampling_ratio/max": 1.8724206686019897, |
| "sampling/importance_sampling_ratio/mean": 0.9904875159263611, |
| "sampling/importance_sampling_ratio/min": 0.527594804763794, |
| "sampling/sampling_logp_difference/max": 0.6394267082214355, |
| "sampling/sampling_logp_difference/mean": 0.017771486192941666, |
| "step": 23, |
| "step_time": 28.308156930143014 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005051148822531104, |
| "clip_ratio/high_mean": 0.0005051148822531104, |
| "clip_ratio/low_mean": 6.318108571576886e-05, |
| "clip_ratio/low_min": 6.318108571576886e-05, |
| "clip_ratio/region_mean": 0.000568295962875709, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2000.0, |
| "completions/mean_length": 1593.7999267578125, |
| "completions/mean_terminated_length": 1264.8966064453125, |
| "completions/min_length": 764.0, |
| "completions/min_terminated_length": 764.0, |
| "entropy": 0.23322015404701232, |
| "epoch": 0.06521739130434782, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017983825877308846, |
| "learning_rate": 5e-05, |
| "loss": 0.1005941778421402, |
| "num_tokens": 2098522.0, |
| "reward": -0.2982226610183716, |
| "reward_std": 0.6998343467712402, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.7782226800918579, |
| "rewards/length_penalty/std": 0.24156226217746735, |
| "sampling/importance_sampling_ratio/max": 1.5202114582061768, |
| "sampling/importance_sampling_ratio/mean": 0.9916720390319824, |
| "sampling/importance_sampling_ratio/min": 0.4922258257865906, |
| "sampling/sampling_logp_difference/max": 0.708817720413208, |
| "sampling/sampling_logp_difference/mean": 0.015332518145442009, |
| "step": 24, |
| "step_time": 27.662194445962086 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007043177334708162, |
| "clip_ratio/high_mean": 0.0007043177334708162, |
| "clip_ratio/low_mean": 0.00014358626067405565, |
| "clip_ratio/low_min": 0.00014358626067405565, |
| "clip_ratio/region_mean": 0.0008479039679514244, |
| "completions/clipped_ratio": 0.2800000011920929, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1861.0, |
| "completions/mean_length": 1399.280029296875, |
| "completions/mean_terminated_length": 1147.0, |
| "completions/min_length": 640.0, |
| "completions/min_terminated_length": 640.0, |
| "entropy": 0.26636312901973724, |
| "epoch": 0.06793478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018543899059295654, |
| "learning_rate": 5e-05, |
| "loss": 0.11409614980220795, |
| "num_tokens": 2170886.0, |
| "reward": 0.056757811456918716, |
| "reward_std": 0.6501639485359192, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.6832422018051147, |
| "rewards/length_penalty/std": 0.24147669970989227, |
| "sampling/importance_sampling_ratio/max": 1.7693860530853271, |
| "sampling/importance_sampling_ratio/mean": 0.9905627965927124, |
| "sampling/importance_sampling_ratio/min": 0.4320516288280487, |
| "sampling/sampling_logp_difference/max": 0.8392102718353271, |
| "sampling/sampling_logp_difference/mean": 0.018011337146162987, |
| "step": 25, |
| "step_time": 26.30519237043336 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008549237565603108, |
| "clip_ratio/high_mean": 0.0008549237565603108, |
| "clip_ratio/low_mean": 0.0001150156487710774, |
| "clip_ratio/low_min": 0.0001150156487710774, |
| "clip_ratio/region_mean": 0.0009699393762275576, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1922.0, |
| "completions/mean_length": 1235.0399169921875, |
| "completions/mean_terminated_length": 1218.448974609375, |
| "completions/min_length": 760.0, |
| "completions/min_terminated_length": 760.0, |
| "entropy": 0.21422497034072877, |
| "epoch": 0.07065217391304347, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018697859719395638, |
| "learning_rate": 5e-05, |
| "loss": 0.10971663147211075, |
| "num_tokens": 2235108.0, |
| "reward": -0.02304687537252903, |
| "reward_std": 0.48166730999946594, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.6030468940734863, |
| "rewards/length_penalty/std": 0.15083222091197968, |
| "sampling/importance_sampling_ratio/max": 1.9734814167022705, |
| "sampling/importance_sampling_ratio/mean": 0.9923763275146484, |
| "sampling/importance_sampling_ratio/min": 0.37951555848121643, |
| "sampling/sampling_logp_difference/max": 0.9688596725463867, |
| "sampling/sampling_logp_difference/mean": 0.015180801041424274, |
| "step": 26, |
| "step_time": 26.001517503987998 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003941807910450734, |
| "clip_ratio/high_mean": 0.0003941807910450734, |
| "clip_ratio/low_mean": 0.00011266354485996998, |
| "clip_ratio/low_min": 0.00011266354485996998, |
| "clip_ratio/region_mean": 0.0005068443395430222, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1952.0, |
| "completions/mean_length": 1497.8800048828125, |
| "completions/mean_terminated_length": 1214.48486328125, |
| "completions/min_length": 626.0, |
| "completions/min_terminated_length": 626.0, |
| "entropy": 0.19425918459892272, |
| "epoch": 0.07336956521739131, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01675463654100895, |
| "learning_rate": 5e-05, |
| "loss": 0.048194363713264465, |
| "num_tokens": 2312252.0, |
| "reward": -0.051386717706918716, |
| "reward_std": 0.6800302863121033, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.7313867211341858, |
| "rewards/length_penalty/std": 0.25491058826446533, |
| "sampling/importance_sampling_ratio/max": 2.0852293968200684, |
| "sampling/importance_sampling_ratio/mean": 0.9928866028785706, |
| "sampling/importance_sampling_ratio/min": 0.32728955149650574, |
| "sampling/sampling_logp_difference/max": 1.1169099807739258, |
| "sampling/sampling_logp_difference/mean": 0.013736417517066002, |
| "step": 27, |
| "step_time": 26.61053773132153 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006696314143482595, |
| "clip_ratio/high_mean": 0.0006696314143482595, |
| "clip_ratio/low_mean": 0.00013978174538351596, |
| "clip_ratio/low_min": 0.00013978174538351596, |
| "clip_ratio/region_mean": 0.0008094131655525417, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2018.0, |
| "completions/max_terminated_length": 2018.0, |
| "completions/mean_length": 1153.6600341796875, |
| "completions/mean_terminated_length": 1153.6600341796875, |
| "completions/min_length": 681.0, |
| "completions/min_terminated_length": 681.0, |
| "entropy": 0.164535990357399, |
| "epoch": 0.07608695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01664278469979763, |
| "learning_rate": 5e-05, |
| "loss": 0.08068578690290451, |
| "num_tokens": 2373295.0, |
| "reward": 0.23668944835662842, |
| "reward_std": 0.45188844203948975, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.5633105635643005, |
| "rewards/length_penalty/std": 0.16880068182945251, |
| "sampling/importance_sampling_ratio/max": 1.907577633857727, |
| "sampling/importance_sampling_ratio/mean": 0.9939589500427246, |
| "sampling/importance_sampling_ratio/min": 0.35874462127685547, |
| "sampling/sampling_logp_difference/max": 1.0251445770263672, |
| "sampling/sampling_logp_difference/mean": 0.012392752803862095, |
| "step": 28, |
| "step_time": 24.809664897387847 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005197454593144357, |
| "clip_ratio/high_mean": 0.0005197454593144357, |
| "clip_ratio/low_mean": 5.26447911397554e-05, |
| "clip_ratio/low_min": 5.26447911397554e-05, |
| "clip_ratio/region_mean": 0.0005723902606405318, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1955.0, |
| "completions/mean_length": 1517.199951171875, |
| "completions/mean_terminated_length": 1243.757568359375, |
| "completions/min_length": 632.0, |
| "completions/min_terminated_length": 632.0, |
| "entropy": 0.22981421351432801, |
| "epoch": 0.07880434782608696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01875111646950245, |
| "learning_rate": 5e-05, |
| "loss": 0.11731194704771042, |
| "num_tokens": 2452595.0, |
| "reward": -0.10082031041383743, |
| "reward_std": 0.6792170405387878, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.7408202886581421, |
| "rewards/length_penalty/std": 0.24844472110271454, |
| "sampling/importance_sampling_ratio/max": 2.2174606323242188, |
| "sampling/importance_sampling_ratio/mean": 0.9917750358581543, |
| "sampling/importance_sampling_ratio/min": 0.3556325137615204, |
| "sampling/sampling_logp_difference/max": 1.0338573455810547, |
| "sampling/sampling_logp_difference/mean": 0.016025636345148087, |
| "step": 29, |
| "step_time": 27.74298789096065 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005712352329283021, |
| "clip_ratio/high_mean": 0.0005712352329283021, |
| "clip_ratio/low_mean": 0.00012854700325988234, |
| "clip_ratio/low_min": 0.00012854700325988234, |
| "clip_ratio/region_mean": 0.000699782240553759, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2024.0, |
| "completions/mean_length": 1216.5999755859375, |
| "completions/mean_terminated_length": 1124.2222900390625, |
| "completions/min_length": 685.0, |
| "completions/min_terminated_length": 685.0, |
| "entropy": 0.16635651588439943, |
| "epoch": 0.08152173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017521338537335396, |
| "learning_rate": 5e-05, |
| "loss": 0.12420433759689331, |
| "num_tokens": 2515215.0, |
| "reward": 0.30595701932907104, |
| "reward_std": 0.46184179186820984, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304574966430664, |
| "rewards/length_penalty/mean": -0.594042956829071, |
| "rewards/length_penalty/std": 0.1965012103319168, |
| "sampling/importance_sampling_ratio/max": 2.539483070373535, |
| "sampling/importance_sampling_ratio/mean": 0.9939422011375427, |
| "sampling/importance_sampling_ratio/min": 0.26190316677093506, |
| "sampling/sampling_logp_difference/max": 1.3397804498672485, |
| "sampling/sampling_logp_difference/mean": 0.013107276521623135, |
| "step": 30, |
| "step_time": 25.327432526275516 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005864212813321501, |
| "clip_ratio/high_mean": 0.0005864212813321501, |
| "clip_ratio/low_mean": 6.815226370235906e-05, |
| "clip_ratio/low_min": 6.815226370235906e-05, |
| "clip_ratio/region_mean": 0.0006545735464897007, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2011.0, |
| "completions/mean_length": 1485.5399169921875, |
| "completions/mean_terminated_length": 1362.0731201171875, |
| "completions/min_length": 562.0, |
| "completions/min_terminated_length": 562.0, |
| "entropy": 0.16528490483760833, |
| "epoch": 0.08423913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017279595136642456, |
| "learning_rate": 5e-05, |
| "loss": 0.10471326112747192, |
| "num_tokens": 2592622.0, |
| "reward": -0.28536131978034973, |
| "reward_std": 0.4804723858833313, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.7253613471984863, |
| "rewards/length_penalty/std": 0.21955737471580505, |
| "sampling/importance_sampling_ratio/max": 2.8586628437042236, |
| "sampling/importance_sampling_ratio/mean": 0.9942410588264465, |
| "sampling/importance_sampling_ratio/min": 0.2360411435365677, |
| "sampling/sampling_logp_difference/max": 1.443749189376831, |
| "sampling/sampling_logp_difference/mean": 0.012665950693190098, |
| "step": 31, |
| "step_time": 27.43080317066051 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008580927504226565, |
| "clip_ratio/high_mean": 0.0008580927504226565, |
| "clip_ratio/low_mean": 3.886102349497378e-05, |
| "clip_ratio/low_min": 3.886102349497378e-05, |
| "clip_ratio/region_mean": 0.0008969537680968642, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1887.0, |
| "completions/mean_length": 1149.719970703125, |
| "completions/mean_terminated_length": 1071.6087646484375, |
| "completions/min_length": 620.0, |
| "completions/min_terminated_length": 620.0, |
| "entropy": 0.22982653677463533, |
| "epoch": 0.08695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019978594034910202, |
| "learning_rate": 5e-05, |
| "loss": 0.14365269243717194, |
| "num_tokens": 2652038.0, |
| "reward": 0.2986132800579071, |
| "reward_std": 0.48640739917755127, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.5613867044448853, |
| "rewards/length_penalty/std": 0.19417373836040497, |
| "sampling/importance_sampling_ratio/max": 2.803338050842285, |
| "sampling/importance_sampling_ratio/mean": 0.9915674328804016, |
| "sampling/importance_sampling_ratio/min": 0.08536958694458008, |
| "sampling/sampling_logp_difference/max": 2.4607653617858887, |
| "sampling/sampling_logp_difference/mean": 0.01730724424123764, |
| "step": 32, |
| "step_time": 25.029227155959234 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006851285259472206, |
| "clip_ratio/high_mean": 0.0006851285259472206, |
| "clip_ratio/low_mean": 0.00010071092401631176, |
| "clip_ratio/low_min": 0.00010071092401631176, |
| "clip_ratio/region_mean": 0.0007858394586946815, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2044.0, |
| "completions/mean_length": 1386.760009765625, |
| "completions/mean_terminated_length": 1177.9473876953125, |
| "completions/min_length": 736.0, |
| "completions/min_terminated_length": 736.0, |
| "entropy": 0.19255209863185882, |
| "epoch": 0.08967391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01860790140926838, |
| "learning_rate": 5e-05, |
| "loss": 0.04447998106479645, |
| "num_tokens": 2725016.0, |
| "reward": -0.037128906697034836, |
| "reward_std": 0.6243671178817749, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732054233551, |
| "rewards/length_penalty/mean": -0.6771289110183716, |
| "rewards/length_penalty/std": 0.2537005543708801, |
| "sampling/importance_sampling_ratio/max": 2.816610336303711, |
| "sampling/importance_sampling_ratio/mean": 0.9930809736251831, |
| "sampling/importance_sampling_ratio/min": 0.2666507661342621, |
| "sampling/sampling_logp_difference/max": 1.3218154907226562, |
| "sampling/sampling_logp_difference/mean": 0.014909989200532436, |
| "step": 33, |
| "step_time": 26.42466412484646 |
| }, |
| { |
| "clip_ratio/high_max": 0.00031742248102091254, |
| "clip_ratio/high_mean": 0.00031742248102091254, |
| "clip_ratio/low_mean": 0.00015698151255492122, |
| "clip_ratio/low_min": 0.00015698151255492122, |
| "clip_ratio/region_mean": 0.0004744039848446846, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1927.0, |
| "completions/mean_length": 1316.8199462890625, |
| "completions/mean_terminated_length": 1134.0250244140625, |
| "completions/min_length": 397.0, |
| "completions/min_terminated_length": 397.0, |
| "entropy": 0.2274552673101425, |
| "epoch": 0.09239130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017942365258932114, |
| "learning_rate": 5e-05, |
| "loss": 0.08828085660934448, |
| "num_tokens": 2794747.0, |
| "reward": 0.1370214819908142, |
| "reward_std": 0.6043517589569092, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.6429784893989563, |
| "rewards/length_penalty/std": 0.22394682466983795, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9917402863502502, |
| "sampling/importance_sampling_ratio/min": 0.2765059471130371, |
| "sampling/sampling_logp_difference/max": 1.2855229377746582, |
| "sampling/sampling_logp_difference/mean": 0.016819080337882042, |
| "step": 34, |
| "step_time": 26.865748363081366 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006646438589086757, |
| "clip_ratio/high_mean": 0.0006646438589086757, |
| "clip_ratio/low_mean": 0.00012726852291962133, |
| "clip_ratio/low_min": 0.00012726852291962133, |
| "clip_ratio/region_mean": 0.0007919123629108072, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2030.0, |
| "completions/mean_length": 1226.0, |
| "completions/mean_terminated_length": 1154.521728515625, |
| "completions/min_length": 714.0, |
| "completions/min_terminated_length": 714.0, |
| "entropy": 0.2424483597278595, |
| "epoch": 0.09510869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022120634093880653, |
| "learning_rate": 5e-05, |
| "loss": 0.1485140472650528, |
| "num_tokens": 2858777.0, |
| "reward": 0.32136717438697815, |
| "reward_std": 0.41920387744903564, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.5986328125, |
| "rewards/length_penalty/std": 0.1873513162136078, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9914340376853943, |
| "sampling/importance_sampling_ratio/min": 0.15862207114696503, |
| "sampling/sampling_logp_difference/max": 1.841230869293213, |
| "sampling/sampling_logp_difference/mean": 0.018726617097854614, |
| "step": 35, |
| "step_time": 25.543159670894966 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005923478078329935, |
| "clip_ratio/high_mean": 0.0005923478078329935, |
| "clip_ratio/low_mean": 0.00016286838508676736, |
| "clip_ratio/low_min": 0.00016286838508676736, |
| "clip_ratio/region_mean": 0.000755216198740527, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1811.0, |
| "completions/mean_length": 1135.0399169921875, |
| "completions/mean_terminated_length": 986.4185791015625, |
| "completions/min_length": 494.0, |
| "completions/min_terminated_length": 494.0, |
| "entropy": 0.2034216493368149, |
| "epoch": 0.09782608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017812756821513176, |
| "learning_rate": 5e-05, |
| "loss": 0.0973254144191742, |
| "num_tokens": 2917999.0, |
| "reward": 0.24578124284744263, |
| "reward_std": 0.6147376298904419, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5542187690734863, |
| "rewards/length_penalty/std": 0.22850991785526276, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9928050637245178, |
| "sampling/importance_sampling_ratio/min": 0.19393190741539001, |
| "sampling/sampling_logp_difference/max": 1.64024817943573, |
| "sampling/sampling_logp_difference/mean": 0.017203593626618385, |
| "step": 36, |
| "step_time": 25.304660103050992 |
| }, |
| { |
| "clip_ratio/high_max": 0.000419435826188419, |
| "clip_ratio/high_mean": 0.000419435826188419, |
| "clip_ratio/low_mean": 0.00011888993612956256, |
| "clip_ratio/low_min": 0.00011888993612956256, |
| "clip_ratio/region_mean": 0.000538325755042024, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1472.0, |
| "completions/mean_length": 1232.0999755859375, |
| "completions/mean_terminated_length": 1001.974365234375, |
| "completions/min_length": 620.0, |
| "completions/min_terminated_length": 620.0, |
| "entropy": 0.23094199895858764, |
| "epoch": 0.10054347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01999955251812935, |
| "learning_rate": 5e-05, |
| "loss": 0.08823837339878082, |
| "num_tokens": 2983124.0, |
| "reward": -0.021611327305436134, |
| "reward_std": 0.6448619365692139, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.601611316204071, |
| "rewards/length_penalty/std": 0.2316407561302185, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9917162656784058, |
| "sampling/importance_sampling_ratio/min": 0.14418137073516846, |
| "sampling/sampling_logp_difference/max": 1.9366832971572876, |
| "sampling/sampling_logp_difference/mean": 0.018293319270014763, |
| "step": 37, |
| "step_time": 26.669016476953402 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004110380803467706, |
| "clip_ratio/high_mean": 0.0004110380803467706, |
| "clip_ratio/low_mean": 0.00016416306025348603, |
| "clip_ratio/low_min": 0.00016416306025348603, |
| "clip_ratio/region_mean": 0.0005752011435106397, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2044.0, |
| "completions/mean_length": 1293.0999755859375, |
| "completions/mean_terminated_length": 1261.6458740234375, |
| "completions/min_length": 714.0, |
| "completions/min_terminated_length": 714.0, |
| "entropy": 0.16276903450489044, |
| "epoch": 0.10326086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01711309142410755, |
| "learning_rate": 5e-05, |
| "loss": 0.07364200800657272, |
| "num_tokens": 3051169.0, |
| "reward": 0.2886035144329071, |
| "reward_std": 0.3920261561870575, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.631396472454071, |
| "rewards/length_penalty/std": 0.23034816980361938, |
| "sampling/importance_sampling_ratio/max": 2.7804486751556396, |
| "sampling/importance_sampling_ratio/mean": 0.9941375851631165, |
| "sampling/importance_sampling_ratio/min": 0.15607857704162598, |
| "sampling/sampling_logp_difference/max": 1.8573956489562988, |
| "sampling/sampling_logp_difference/mean": 0.013436587527394295, |
| "step": 38, |
| "step_time": 25.98156057903543 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006271645077504217, |
| "clip_ratio/high_mean": 0.0006271645077504217, |
| "clip_ratio/low_mean": 0.00010634372592903674, |
| "clip_ratio/low_min": 0.00010634372592903674, |
| "clip_ratio/region_mean": 0.0007335082162171602, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1754.0, |
| "completions/mean_length": 1173.780029296875, |
| "completions/mean_terminated_length": 955.2250366210938, |
| "completions/min_length": 581.0, |
| "completions/min_terminated_length": 581.0, |
| "entropy": 0.18645241856575012, |
| "epoch": 0.10597826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018314942717552185, |
| "learning_rate": 5e-05, |
| "loss": 0.06761015951633453, |
| "num_tokens": 3112348.0, |
| "reward": 0.04686523228883743, |
| "reward_std": 0.7094346880912781, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.5731347799301147, |
| "rewards/length_penalty/std": 0.25952664017677307, |
| "sampling/importance_sampling_ratio/max": 2.629316568374634, |
| "sampling/importance_sampling_ratio/mean": 0.9932643175125122, |
| "sampling/importance_sampling_ratio/min": 0.1535349190235138, |
| "sampling/sampling_logp_difference/max": 1.8738272190093994, |
| "sampling/sampling_logp_difference/mean": 0.01502909418195486, |
| "step": 39, |
| "step_time": 25.790251429891214 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009092503576539456, |
| "clip_ratio/high_mean": 0.0009092503576539456, |
| "clip_ratio/low_mean": 8.635809645056725e-05, |
| "clip_ratio/low_min": 8.635809645056725e-05, |
| "clip_ratio/region_mean": 0.000995608454104513, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1440.0, |
| "completions/max_terminated_length": 1440.0, |
| "completions/mean_length": 870.3399658203125, |
| "completions/mean_terminated_length": 870.3399658203125, |
| "completions/min_length": 618.0, |
| "completions/min_terminated_length": 618.0, |
| "entropy": 0.18192007541656494, |
| "epoch": 0.10869565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01616382785141468, |
| "learning_rate": 5e-05, |
| "loss": 0.06506504118442535, |
| "num_tokens": 3158585.0, |
| "reward": 0.5750293135643005, |
| "reward_std": 0.0848696380853653, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.42497071623802185, |
| "rewards/length_penalty/std": 0.0848696380853653, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9931731224060059, |
| "sampling/importance_sampling_ratio/min": 0.11430167406797409, |
| "sampling/sampling_logp_difference/max": 2.1689140796661377, |
| "sampling/sampling_logp_difference/mean": 0.017184380441904068, |
| "step": 40, |
| "step_time": 17.641971144126728 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006591534824110567, |
| "clip_ratio/high_mean": 0.0006591534824110567, |
| "clip_ratio/low_mean": 3.45005581039004e-05, |
| "clip_ratio/low_min": 3.45005581039004e-05, |
| "clip_ratio/region_mean": 0.0006936540419701486, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1753.0, |
| "completions/mean_length": 1133.5599365234375, |
| "completions/mean_terminated_length": 1095.4583740234375, |
| "completions/min_length": 565.0, |
| "completions/min_terminated_length": 565.0, |
| "entropy": 0.19037298262119293, |
| "epoch": 0.11141304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018935466185212135, |
| "learning_rate": 5e-05, |
| "loss": 0.07612141966819763, |
| "num_tokens": 3219003.0, |
| "reward": -0.013496093451976776, |
| "reward_std": 0.5555275678634644, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.5534961223602295, |
| "rewards/length_penalty/std": 0.16630038619041443, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9930751919746399, |
| "sampling/importance_sampling_ratio/min": 0.1459389477968216, |
| "sampling/sampling_logp_difference/max": 1.9245668649673462, |
| "sampling/sampling_logp_difference/mean": 0.016215359792113304, |
| "step": 41, |
| "step_time": 25.54675062862225 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007439341745339334, |
| "clip_ratio/high_mean": 0.0007439341745339334, |
| "clip_ratio/low_mean": 0.00013028232351643964, |
| "clip_ratio/low_min": 0.00013028232351643964, |
| "clip_ratio/region_mean": 0.0008742164936847985, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1986.0, |
| "completions/mean_length": 1145.179931640625, |
| "completions/mean_terminated_length": 998.2092895507812, |
| "completions/min_length": 453.0, |
| "completions/min_terminated_length": 453.0, |
| "entropy": 0.18841493129730225, |
| "epoch": 0.11413043478260869, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019623132422566414, |
| "learning_rate": 5e-05, |
| "loss": 0.07725121080875397, |
| "num_tokens": 3279252.0, |
| "reward": 0.30083006620407104, |
| "reward_std": 0.5640762448310852, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.5591699481010437, |
| "rewards/length_penalty/std": 0.26338833570480347, |
| "sampling/importance_sampling_ratio/max": 2.1462347507476807, |
| "sampling/importance_sampling_ratio/mean": 0.9933086037635803, |
| "sampling/importance_sampling_ratio/min": 0.15205901861190796, |
| "sampling/sampling_logp_difference/max": 1.8834865093231201, |
| "sampling/sampling_logp_difference/mean": 0.016336476430296898, |
| "step": 42, |
| "step_time": 25.846705763135105 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006595640617888421, |
| "clip_ratio/high_mean": 0.0006595640617888421, |
| "clip_ratio/low_mean": 0.00011880509555339813, |
| "clip_ratio/low_min": 0.00011880509555339813, |
| "clip_ratio/region_mean": 0.0007783691631630063, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1704.0, |
| "completions/max_terminated_length": 1704.0, |
| "completions/mean_length": 812.8599853515625, |
| "completions/mean_terminated_length": 812.8599853515625, |
| "completions/min_length": 445.0, |
| "completions/min_terminated_length": 445.0, |
| "entropy": 0.20379752516746522, |
| "epoch": 0.11684782608695653, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016623755916953087, |
| "learning_rate": 5e-05, |
| "loss": 0.09463626146316528, |
| "num_tokens": 3321865.0, |
| "reward": 0.6030957102775574, |
| "reward_std": 0.15212664008140564, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.3969042897224426, |
| "rewards/length_penalty/std": 0.15212664008140564, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9928601384162903, |
| "sampling/importance_sampling_ratio/min": 0.15553829073905945, |
| "sampling/sampling_logp_difference/max": 1.8608633279800415, |
| "sampling/sampling_logp_difference/mean": 0.019999099895358086, |
| "step": 43, |
| "step_time": 19.98531307396479 |
| }, |
| { |
| "clip_ratio/high_max": 0.000692727358546108, |
| "clip_ratio/high_mean": 0.000692727358546108, |
| "clip_ratio/low_mean": 0.0001292879198445007, |
| "clip_ratio/low_min": 0.0001292879198445007, |
| "clip_ratio/region_mean": 0.0008220152638386935, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1998.0, |
| "completions/mean_length": 1104.7999267578125, |
| "completions/mean_terminated_length": 1065.5, |
| "completions/min_length": 478.0, |
| "completions/min_terminated_length": 478.0, |
| "entropy": 0.19660083651542665, |
| "epoch": 0.11956521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01956993155181408, |
| "learning_rate": 5e-05, |
| "loss": 0.08270780742168427, |
| "num_tokens": 3380105.0, |
| "reward": 0.42054685950279236, |
| "reward_std": 0.35633406043052673, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.5394531488418579, |
| "rewards/length_penalty/std": 0.22404155135154724, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.992460310459137, |
| "sampling/importance_sampling_ratio/min": 0.14695732295513153, |
| "sampling/sampling_logp_difference/max": 1.9176130294799805, |
| "sampling/sampling_logp_difference/mean": 0.017957350239157677, |
| "step": 44, |
| "step_time": 25.09302610415034 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009103513788431883, |
| "clip_ratio/high_mean": 0.0009103513788431883, |
| "clip_ratio/low_mean": 5.670995160471648e-05, |
| "clip_ratio/low_min": 5.670995160471648e-05, |
| "clip_ratio/region_mean": 0.0009670613217167556, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1971.0, |
| "completions/mean_length": 1118.6400146484375, |
| "completions/mean_terminated_length": 991.9091186523438, |
| "completions/min_length": 467.0, |
| "completions/min_terminated_length": 467.0, |
| "entropy": 0.21414778530597686, |
| "epoch": 0.12228260869565218, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02317577414214611, |
| "learning_rate": 5e-05, |
| "loss": 0.11955960839986801, |
| "num_tokens": 3438977.0, |
| "reward": 0.1937890648841858, |
| "reward_std": 0.6458888053894043, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.5462109446525574, |
| "rewards/length_penalty/std": 0.26540839672088623, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9921079277992249, |
| "sampling/importance_sampling_ratio/min": 0.1596342772245407, |
| "sampling/sampling_logp_difference/max": 2.1383285522460938, |
| "sampling/sampling_logp_difference/mean": 0.018177490681409836, |
| "step": 45, |
| "step_time": 25.18644618219696 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012181638274341821, |
| "clip_ratio/high_mean": 0.0012181638274341821, |
| "clip_ratio/low_mean": 7.497181795770302e-05, |
| "clip_ratio/low_min": 7.497181795770302e-05, |
| "clip_ratio/region_mean": 0.0012931356206536293, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1226.0, |
| "completions/mean_length": 1089.02001953125, |
| "completions/mean_terminated_length": 849.2750244140625, |
| "completions/min_length": 454.0, |
| "completions/min_terminated_length": 454.0, |
| "entropy": 0.22630111277103424, |
| "epoch": 0.125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016400394961237907, |
| "learning_rate": 5e-05, |
| "loss": 0.042683668434619904, |
| "num_tokens": 3496938.0, |
| "reward": 0.2682519555091858, |
| "reward_std": 0.6458777189254761, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5317480564117432, |
| "rewards/length_penalty/std": 0.25053834915161133, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9916934370994568, |
| "sampling/importance_sampling_ratio/min": 0.08636664599180222, |
| "sampling/sampling_logp_difference/max": 2.4491536617279053, |
| "sampling/sampling_logp_difference/mean": 0.01941664330661297, |
| "step": 46, |
| "step_time": 25.717923732707277 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007034850481431931, |
| "clip_ratio/high_mean": 0.0007034850481431931, |
| "clip_ratio/low_mean": 0.0001471527459216304, |
| "clip_ratio/low_min": 0.0001471527459216304, |
| "clip_ratio/region_mean": 0.0008506378042511642, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1988.0, |
| "completions/mean_length": 919.1399536132812, |
| "completions/mean_terminated_length": 847.0850830078125, |
| "completions/min_length": 525.0, |
| "completions/min_terminated_length": 525.0, |
| "entropy": 0.20610004663467407, |
| "epoch": 0.12771739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01724245771765709, |
| "learning_rate": 5e-05, |
| "loss": 0.05881120637059212, |
| "num_tokens": 3545285.0, |
| "reward": 0.4512011706829071, |
| "reward_std": 0.4983653426170349, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304577946662903, |
| "rewards/length_penalty/mean": -0.4487988352775574, |
| "rewards/length_penalty/std": 0.22889265418052673, |
| "sampling/importance_sampling_ratio/max": 2.5950088500976562, |
| "sampling/importance_sampling_ratio/mean": 0.9923014044761658, |
| "sampling/importance_sampling_ratio/min": 0.10353942960500717, |
| "sampling/sampling_logp_difference/max": 2.2678027153015137, |
| "sampling/sampling_logp_difference/mean": 0.018676703795790672, |
| "step": 47, |
| "step_time": 24.33197053289041 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007297933771042153, |
| "clip_ratio/high_mean": 0.0007297933771042153, |
| "clip_ratio/low_mean": 3.783966094488278e-05, |
| "clip_ratio/low_min": 3.783966094488278e-05, |
| "clip_ratio/region_mean": 0.0007676330336835235, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2031.0, |
| "completions/mean_length": 1046.1199951171875, |
| "completions/mean_terminated_length": 934.800048828125, |
| "completions/min_length": 437.0, |
| "completions/min_terminated_length": 437.0, |
| "entropy": 0.16849468648433685, |
| "epoch": 0.13043478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0175967738032341, |
| "learning_rate": 5e-05, |
| "loss": 0.04348339885473251, |
| "num_tokens": 3600551.0, |
| "reward": 0.1491992175579071, |
| "reward_std": 0.6337429285049438, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.5108007788658142, |
| "rewards/length_penalty/std": 0.2482379525899887, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9937344193458557, |
| "sampling/importance_sampling_ratio/min": 0.11326346546411514, |
| "sampling/sampling_logp_difference/max": 2.1780385971069336, |
| "sampling/sampling_logp_difference/mean": 0.016115352511405945, |
| "step": 48, |
| "step_time": 24.871384381316602 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010520221141632647, |
| "clip_ratio/high_mean": 0.0010520221141632647, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010520221141632647, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1221.0, |
| "completions/mean_length": 978.3399658203125, |
| "completions/mean_terminated_length": 710.9249877929688, |
| "completions/min_length": 369.0, |
| "completions/min_terminated_length": 369.0, |
| "entropy": 0.23513416349887847, |
| "epoch": 0.1331521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014279501512646675, |
| "learning_rate": 5e-05, |
| "loss": 0.045435212552547455, |
| "num_tokens": 3652158.0, |
| "reward": 0.3222949206829071, |
| "reward_std": 0.6740695238113403, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.47770509123802185, |
| "rewards/length_penalty/std": 0.27914658188819885, |
| "sampling/importance_sampling_ratio/max": 2.6580207347869873, |
| "sampling/importance_sampling_ratio/mean": 0.9915663003921509, |
| "sampling/importance_sampling_ratio/min": 0.08653713017702103, |
| "sampling/sampling_logp_difference/max": 2.4471817016601562, |
| "sampling/sampling_logp_difference/mean": 0.020803218707442284, |
| "step": 49, |
| "step_time": 25.104636664967984 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015973969304468483, |
| "clip_ratio/high_mean": 0.0015973969304468483, |
| "clip_ratio/low_mean": 1.891074061859399e-05, |
| "clip_ratio/low_min": 1.891074061859399e-05, |
| "clip_ratio/region_mean": 0.0016163076681550593, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1039.0, |
| "completions/mean_length": 933.0, |
| "completions/mean_terminated_length": 654.25, |
| "completions/min_length": 282.0, |
| "completions/min_terminated_length": 282.0, |
| "entropy": 0.24436002373695373, |
| "epoch": 0.1358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013328667730093002, |
| "learning_rate": 5e-05, |
| "loss": 0.04158805310726166, |
| "num_tokens": 3701218.0, |
| "reward": 0.14443358778953552, |
| "reward_std": 0.7250430583953857, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.45556640625, |
| "rewards/length_penalty/std": 0.2883836627006531, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9908113479614258, |
| "sampling/importance_sampling_ratio/min": 0.04097478464245796, |
| "sampling/sampling_logp_difference/max": 3.194798469543457, |
| "sampling/sampling_logp_difference/mean": 0.02187870815396309, |
| "step": 50, |
| "step_time": 24.471280077006668 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009731275727972388, |
| "clip_ratio/high_mean": 0.0009731275727972388, |
| "clip_ratio/low_mean": 7.188716117525473e-05, |
| "clip_ratio/low_min": 7.188716117525473e-05, |
| "clip_ratio/region_mean": 0.0010450147150550038, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1480.0, |
| "completions/max_terminated_length": 1480.0, |
| "completions/mean_length": 818.239990234375, |
| "completions/mean_terminated_length": 818.239990234375, |
| "completions/min_length": 426.0, |
| "completions/min_terminated_length": 426.0, |
| "entropy": 0.17211617529392242, |
| "epoch": 0.13858695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014803892001509666, |
| "learning_rate": 5e-05, |
| "loss": 0.04142213612794876, |
| "num_tokens": 3744640.0, |
| "reward": 0.5804687142372131, |
| "reward_std": 0.20210698246955872, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.3995312452316284, |
| "rewards/length_penalty/std": 0.13855627179145813, |
| "sampling/importance_sampling_ratio/max": 2.7023584842681885, |
| "sampling/importance_sampling_ratio/mean": 0.9934318661689758, |
| "sampling/importance_sampling_ratio/min": 0.12204709649085999, |
| "sampling/sampling_logp_difference/max": 2.1033482551574707, |
| "sampling/sampling_logp_difference/mean": 0.018068784847855568, |
| "step": 51, |
| "step_time": 17.94456404563971 |
| }, |
| { |
| "clip_ratio/high_max": 0.00053445250086952, |
| "clip_ratio/high_mean": 0.00053445250086952, |
| "clip_ratio/low_mean": 9.609094267943874e-05, |
| "clip_ratio/low_min": 9.609094267943874e-05, |
| "clip_ratio/region_mean": 0.0006305434450041503, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1927.0, |
| "completions/mean_length": 792.9199829101562, |
| "completions/mean_terminated_length": 767.3060913085938, |
| "completions/min_length": 387.0, |
| "completions/min_terminated_length": 387.0, |
| "entropy": 0.168829745054245, |
| "epoch": 0.14130434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019603334367275238, |
| "learning_rate": 5e-05, |
| "loss": 0.07455220818519592, |
| "num_tokens": 3786546.0, |
| "reward": 0.3928320109844208, |
| "reward_std": 0.4633483290672302, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.38716796040534973, |
| "rewards/length_penalty/std": 0.16155347228050232, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9938976764678955, |
| "sampling/importance_sampling_ratio/min": 0.06568993628025055, |
| "sampling/sampling_logp_difference/max": 2.7228095531463623, |
| "sampling/sampling_logp_difference/mean": 0.018862249329686165, |
| "step": 52, |
| "step_time": 24.00628535822034 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006544351446791552, |
| "clip_ratio/high_mean": 0.0006544351446791552, |
| "clip_ratio/low_mean": 9.588491375325247e-05, |
| "clip_ratio/low_min": 9.588491375325247e-05, |
| "clip_ratio/region_mean": 0.0007503200526116416, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1401.0, |
| "completions/mean_length": 871.0199584960938, |
| "completions/mean_terminated_length": 795.8936157226562, |
| "completions/min_length": 376.0, |
| "completions/min_terminated_length": 376.0, |
| "entropy": 0.19160565733909607, |
| "epoch": 0.14402173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02219633385539055, |
| "learning_rate": 5e-05, |
| "loss": 0.12224940210580826, |
| "num_tokens": 3832817.0, |
| "reward": 0.314697265625, |
| "reward_std": 0.606342077255249, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.42530274391174316, |
| "rewards/length_penalty/std": 0.19953002035617828, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9928863644599915, |
| "sampling/importance_sampling_ratio/min": 0.05365324765443802, |
| "sampling/sampling_logp_difference/max": 2.92521333694458, |
| "sampling/sampling_logp_difference/mean": 0.020148755982518196, |
| "step": 53, |
| "step_time": 24.095060840947554 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004004340240499005, |
| "clip_ratio/high_mean": 0.0004004340240499005, |
| "clip_ratio/low_mean": 8.419621444772929e-05, |
| "clip_ratio/low_min": 8.419621444772929e-05, |
| "clip_ratio/region_mean": 0.00048463023849762974, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1965.0, |
| "completions/mean_length": 1204.5999755859375, |
| "completions/mean_terminated_length": 1067.3023681640625, |
| "completions/min_length": 566.0, |
| "completions/min_terminated_length": 566.0, |
| "entropy": 0.21263552606105804, |
| "epoch": 0.14673913043478262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018590528517961502, |
| "learning_rate": 5e-05, |
| "loss": 0.09691454470157623, |
| "num_tokens": 3896867.0, |
| "reward": 0.21181640028953552, |
| "reward_std": 0.6058595776557922, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.588183581829071, |
| "rewards/length_penalty/std": 0.25641068816185, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9919548034667969, |
| "sampling/importance_sampling_ratio/min": 0.060499291867017746, |
| "sampling/sampling_logp_difference/max": 2.8051235675811768, |
| "sampling/sampling_logp_difference/mean": 0.01935158111155033, |
| "step": 54, |
| "step_time": 25.598680251277983 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008707194036105647, |
| "clip_ratio/high_mean": 0.0008707194036105647, |
| "clip_ratio/low_mean": 0.00010068186529679223, |
| "clip_ratio/low_min": 0.00010068186529679223, |
| "clip_ratio/region_mean": 0.0009714012674521654, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1948.0, |
| "completions/mean_length": 1096.919921875, |
| "completions/mean_terminated_length": 967.227294921875, |
| "completions/min_length": 448.0, |
| "completions/min_terminated_length": 448.0, |
| "entropy": 0.22216415405273438, |
| "epoch": 0.14945652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021540069952607155, |
| "learning_rate": 5e-05, |
| "loss": 0.094159334897995, |
| "num_tokens": 3954603.0, |
| "reward": 0.26439452171325684, |
| "reward_std": 0.6379780769348145, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5356054902076721, |
| "rewards/length_penalty/std": 0.27845966815948486, |
| "sampling/importance_sampling_ratio/max": 2.7441539764404297, |
| "sampling/importance_sampling_ratio/mean": 0.9917101263999939, |
| "sampling/importance_sampling_ratio/min": 0.08549530804157257, |
| "sampling/sampling_logp_difference/max": 2.459293842315674, |
| "sampling/sampling_logp_difference/mean": 0.020652635022997856, |
| "step": 55, |
| "step_time": 25.68535593058914 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009710750076919794, |
| "clip_ratio/high_mean": 0.0009710750076919794, |
| "clip_ratio/low_mean": 7.646471203770489e-05, |
| "clip_ratio/low_min": 7.646471203770489e-05, |
| "clip_ratio/region_mean": 0.0010475397109985351, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1251.0, |
| "completions/mean_length": 922.3200073242188, |
| "completions/mean_terminated_length": 640.9000244140625, |
| "completions/min_length": 360.0, |
| "completions/min_terminated_length": 360.0, |
| "entropy": 0.18780343532562255, |
| "epoch": 0.15217391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014794361777603626, |
| "learning_rate": 5e-05, |
| "loss": 0.024544673040509224, |
| "num_tokens": 4003789.0, |
| "reward": 0.12964843213558197, |
| "reward_std": 0.6917726397514343, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.4503515660762787, |
| "rewards/length_penalty/std": 0.29927679896354675, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9923670291900635, |
| "sampling/importance_sampling_ratio/min": 0.0570797398686409, |
| "sampling/sampling_logp_difference/max": 2.8633060455322266, |
| "sampling/sampling_logp_difference/mean": 0.018756825476884842, |
| "step": 56, |
| "step_time": 24.448076013242826 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004333902965299785, |
| "clip_ratio/high_mean": 0.0004333902965299785, |
| "clip_ratio/low_mean": 0.0001851662207627669, |
| "clip_ratio/low_min": 0.0001851662207627669, |
| "clip_ratio/region_mean": 0.0006185565143823624, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1776.0, |
| "completions/mean_length": 1099.4599609375, |
| "completions/mean_terminated_length": 799.9210815429688, |
| "completions/min_length": 347.0, |
| "completions/min_terminated_length": 347.0, |
| "entropy": 0.26923188418149946, |
| "epoch": 0.15489130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01690416783094406, |
| "learning_rate": 5e-05, |
| "loss": 0.05235477164387703, |
| "num_tokens": 4062002.0, |
| "reward": 0.22315429151058197, |
| "reward_std": 0.7157526016235352, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.5368456840515137, |
| "rewards/length_penalty/std": 0.3151639401912689, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9897788166999817, |
| "sampling/importance_sampling_ratio/min": 0.13390783965587616, |
| "sampling/sampling_logp_difference/max": 2.010603427886963, |
| "sampling/sampling_logp_difference/mean": 0.022602049633860588, |
| "step": 57, |
| "step_time": 25.177973021985963 |
| }, |
| { |
| "clip_ratio/high_max": 0.000699937401805073, |
| "clip_ratio/high_mean": 0.000699937401805073, |
| "clip_ratio/low_mean": 0.00010630405740812421, |
| "clip_ratio/low_min": 0.00010630405740812421, |
| "clip_ratio/region_mean": 0.0008062414592131973, |
| "completions/clipped_ratio": 0.25999999046325684, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1704.0, |
| "completions/mean_length": 1233.8599853515625, |
| "completions/mean_terminated_length": 947.8108520507812, |
| "completions/min_length": 428.0, |
| "completions/min_terminated_length": 428.0, |
| "entropy": 0.24144376814365387, |
| "epoch": 0.15760869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021203303709626198, |
| "learning_rate": 5e-05, |
| "loss": 0.09330578148365021, |
| "num_tokens": 4127505.0, |
| "reward": -0.0024707030970603228, |
| "reward_std": 0.7344722151756287, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.6024706959724426, |
| "rewards/length_penalty/std": 0.2935793697834015, |
| "sampling/importance_sampling_ratio/max": 2.673166275024414, |
| "sampling/importance_sampling_ratio/mean": 0.9907755255699158, |
| "sampling/importance_sampling_ratio/min": 0.07911914587020874, |
| "sampling/sampling_logp_difference/max": 2.5368003845214844, |
| "sampling/sampling_logp_difference/mean": 0.021435290575027466, |
| "step": 58, |
| "step_time": 26.790618495782837 |
| }, |
| { |
| "clip_ratio/high_max": 0.00043638584902510046, |
| "clip_ratio/high_mean": 0.00043638584902510046, |
| "clip_ratio/low_mean": 0.00010769373475341126, |
| "clip_ratio/low_min": 0.00010769373475341126, |
| "clip_ratio/region_mean": 0.0005440795910544693, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2001.0, |
| "completions/mean_length": 800.5199584960938, |
| "completions/mean_terminated_length": 748.5416870117188, |
| "completions/min_length": 359.0, |
| "completions/min_terminated_length": 359.0, |
| "entropy": 0.17738782465457917, |
| "epoch": 0.16032608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021261993795633316, |
| "learning_rate": 5e-05, |
| "loss": 0.10207454860210419, |
| "num_tokens": 4170241.0, |
| "reward": 0.48912107944488525, |
| "reward_std": 0.47969743609428406, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826071977615356, |
| "rewards/length_penalty/mean": -0.39087891578674316, |
| "rewards/length_penalty/std": 0.20807500183582306, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9926632642745972, |
| "sampling/importance_sampling_ratio/min": 0.05151674151420593, |
| "sampling/sampling_logp_difference/max": 2.965848445892334, |
| "sampling/sampling_logp_difference/mean": 0.020642835646867752, |
| "step": 59, |
| "step_time": 23.709213282214478 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002563061600085348, |
| "clip_ratio/high_mean": 0.0002563061600085348, |
| "clip_ratio/low_mean": 0.00010330905206501484, |
| "clip_ratio/low_min": 0.00010330905206501484, |
| "clip_ratio/region_mean": 0.00035961521789431574, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1233.0, |
| "completions/max_terminated_length": 1233.0, |
| "completions/mean_length": 772.4400024414062, |
| "completions/mean_terminated_length": 772.4400024414062, |
| "completions/min_length": 392.0, |
| "completions/min_terminated_length": 392.0, |
| "entropy": 0.14034459441900254, |
| "epoch": 0.16304347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01492391712963581, |
| "learning_rate": 5e-05, |
| "loss": 0.04165235906839371, |
| "num_tokens": 4211673.0, |
| "reward": 0.002832031110301614, |
| "reward_std": 0.5176273584365845, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.3771679699420929, |
| "rewards/length_penalty/std": 0.11115802824497223, |
| "sampling/importance_sampling_ratio/max": 2.4209439754486084, |
| "sampling/importance_sampling_ratio/mean": 0.9949809908866882, |
| "sampling/importance_sampling_ratio/min": 0.051577843725681305, |
| "sampling/sampling_logp_difference/max": 2.964663028717041, |
| "sampling/sampling_logp_difference/mean": 0.01739756390452385, |
| "step": 60, |
| "step_time": 15.429080261848867 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007448806398315355, |
| "clip_ratio/high_mean": 0.0007448806398315355, |
| "clip_ratio/low_mean": 2.9797377646900713e-05, |
| "clip_ratio/low_min": 2.9797377646900713e-05, |
| "clip_ratio/region_mean": 0.0007746780174784362, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 911.0, |
| "completions/max_terminated_length": 911.0, |
| "completions/mean_length": 645.97998046875, |
| "completions/mean_terminated_length": 645.97998046875, |
| "completions/min_length": 400.0, |
| "completions/min_terminated_length": 400.0, |
| "entropy": 0.14809348285198212, |
| "epoch": 0.16576086956521738, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014203577302396297, |
| "learning_rate": 5e-05, |
| "loss": 0.038087740540504456, |
| "num_tokens": 4246382.0, |
| "reward": 0.6845800876617432, |
| "reward_std": 0.05958530306816101, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.31541991233825684, |
| "rewards/length_penalty/std": 0.05958529934287071, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9946348071098328, |
| "sampling/importance_sampling_ratio/min": 0.037354156374931335, |
| "sampling/sampling_logp_difference/max": 3.28731107711792, |
| "sampling/sampling_logp_difference/mean": 0.020090077072381973, |
| "step": 61, |
| "step_time": 11.506354367826134 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008136456366628408, |
| "clip_ratio/high_mean": 0.0008136456366628408, |
| "clip_ratio/low_mean": 3.931597457267344e-05, |
| "clip_ratio/low_min": 3.931597457267344e-05, |
| "clip_ratio/region_mean": 0.0008529616097803228, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1930.0, |
| "completions/mean_length": 940.2999877929688, |
| "completions/mean_terminated_length": 917.69384765625, |
| "completions/min_length": 289.0, |
| "completions/min_terminated_length": 289.0, |
| "entropy": 0.16105791926383972, |
| "epoch": 0.16847826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018179485574364662, |
| "learning_rate": 5e-05, |
| "loss": 0.06390588730573654, |
| "num_tokens": 4297017.0, |
| "reward": 0.34086912870407104, |
| "reward_std": 0.49857690930366516, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.4591308534145355, |
| "rewards/length_penalty/std": 0.1968691498041153, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.993736743927002, |
| "sampling/importance_sampling_ratio/min": 0.09221337735652924, |
| "sampling/sampling_logp_difference/max": 2.383650064468384, |
| "sampling/sampling_logp_difference/mean": 0.01728961057960987, |
| "step": 62, |
| "step_time": 24.922941673081368 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004660156613681465, |
| "clip_ratio/high_mean": 0.0004660156613681465, |
| "clip_ratio/low_mean": 5.673716950695962e-05, |
| "clip_ratio/low_min": 5.673716950695962e-05, |
| "clip_ratio/region_mean": 0.0005227528279647231, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1615.0, |
| "completions/max_terminated_length": 1615.0, |
| "completions/mean_length": 731.0399780273438, |
| "completions/mean_terminated_length": 731.0399780273438, |
| "completions/min_length": 244.0, |
| "completions/min_terminated_length": 244.0, |
| "entropy": 0.1496315360069275, |
| "epoch": 0.17119565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014387847855687141, |
| "learning_rate": 5e-05, |
| "loss": 0.04436701908707619, |
| "num_tokens": 4336819.0, |
| "reward": 0.4430468678474426, |
| "reward_std": 0.5486912727355957, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.35695311427116394, |
| "rewards/length_penalty/std": 0.17223653197288513, |
| "sampling/importance_sampling_ratio/max": 2.5042917728424072, |
| "sampling/importance_sampling_ratio/mean": 0.994520902633667, |
| "sampling/importance_sampling_ratio/min": 0.06114684417843819, |
| "sampling/sampling_logp_difference/max": 2.7944769859313965, |
| "sampling/sampling_logp_difference/mean": 0.01874612830579281, |
| "step": 63, |
| "step_time": 18.922958778217435 |
| }, |
| { |
| "clip_ratio/high_max": 0.000792847282718867, |
| "clip_ratio/high_mean": 0.000792847282718867, |
| "clip_ratio/low_mean": 0.0001386634394293651, |
| "clip_ratio/low_min": 0.0001386634394293651, |
| "clip_ratio/region_mean": 0.0009315107250586152, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1464.0, |
| "completions/mean_length": 750.7999877929688, |
| "completions/mean_terminated_length": 724.3265380859375, |
| "completions/min_length": 334.0, |
| "completions/min_terminated_length": 334.0, |
| "entropy": 0.1570049375295639, |
| "epoch": 0.17391304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020466351881623268, |
| "learning_rate": 5e-05, |
| "loss": 0.08620959520339966, |
| "num_tokens": 4377879.0, |
| "reward": 0.5133984088897705, |
| "reward_std": 0.38893499970436096, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826071977615356, |
| "rewards/length_penalty/mean": -0.3666015565395355, |
| "rewards/length_penalty/std": 0.13564565777778625, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9939337372779846, |
| "sampling/importance_sampling_ratio/min": 0.16536609828472137, |
| "sampling/sampling_logp_difference/max": 1.7995935678482056, |
| "sampling/sampling_logp_difference/mean": 0.01920284703373909, |
| "step": 64, |
| "step_time": 22.34448242606595 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005973302468191832, |
| "clip_ratio/high_mean": 0.0005973302468191832, |
| "clip_ratio/low_mean": 7.574164192192256e-05, |
| "clip_ratio/low_min": 7.574164192192256e-05, |
| "clip_ratio/region_mean": 0.0006730718887411058, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1752.0, |
| "completions/mean_length": 821.6199951171875, |
| "completions/mean_terminated_length": 770.5208740234375, |
| "completions/min_length": 378.0, |
| "completions/min_terminated_length": 378.0, |
| "entropy": 0.1671563982963562, |
| "epoch": 0.1766304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017389990389347076, |
| "learning_rate": 5e-05, |
| "loss": 0.08777762949466705, |
| "num_tokens": 4422670.0, |
| "reward": 0.5188183188438416, |
| "reward_std": 0.43460097908973694, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.4011816382408142, |
| "rewards/length_penalty/std": 0.18575416505336761, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9935203194618225, |
| "sampling/importance_sampling_ratio/min": 0.06127934902906418, |
| "sampling/sampling_logp_difference/max": 2.7923123836517334, |
| "sampling/sampling_logp_difference/mean": 0.019455773755908012, |
| "step": 65, |
| "step_time": 22.99565628520213 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005780523584689945, |
| "clip_ratio/high_mean": 0.0005780523584689945, |
| "clip_ratio/low_mean": 0.0002622720610816032, |
| "clip_ratio/low_min": 0.0002622720610816032, |
| "clip_ratio/region_mean": 0.0008403244195505976, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1996.0, |
| "completions/mean_length": 791.3200073242188, |
| "completions/mean_terminated_length": 586.7442016601562, |
| "completions/min_length": 295.0, |
| "completions/min_terminated_length": 295.0, |
| "entropy": 0.26058044135570524, |
| "epoch": 0.1793478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017154205590486526, |
| "learning_rate": 5e-05, |
| "loss": 0.04564708471298218, |
| "num_tokens": 4465286.0, |
| "reward": 0.37361326813697815, |
| "reward_std": 0.6931055784225464, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.3863867223262787, |
| "rewards/length_penalty/std": 0.28513047099113464, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9893307089805603, |
| "sampling/importance_sampling_ratio/min": 0.1172771081328392, |
| "sampling/sampling_logp_difference/max": 2.1432156562805176, |
| "sampling/sampling_logp_difference/mean": 0.026307187974452972, |
| "step": 66, |
| "step_time": 23.517513233004138 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006526641664095223, |
| "clip_ratio/high_mean": 0.0006526641664095223, |
| "clip_ratio/low_mean": 0.00013707599136978388, |
| "clip_ratio/low_min": 0.00013707599136978388, |
| "clip_ratio/region_mean": 0.0007897401577793062, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1453.0, |
| "completions/max_terminated_length": 1453.0, |
| "completions/mean_length": 731.4199829101562, |
| "completions/mean_terminated_length": 731.4199829101562, |
| "completions/min_length": 213.0, |
| "completions/min_terminated_length": 213.0, |
| "entropy": 0.1294546380639076, |
| "epoch": 0.18206521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01403296273201704, |
| "learning_rate": 5e-05, |
| "loss": 0.042149074375629425, |
| "num_tokens": 4504807.0, |
| "reward": 0.2428613156080246, |
| "reward_std": 0.5500320196151733, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.35713866353034973, |
| "rewards/length_penalty/std": 0.15154975652694702, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9952972531318665, |
| "sampling/importance_sampling_ratio/min": 0.10496558248996735, |
| "sampling/sampling_logp_difference/max": 2.254122734069824, |
| "sampling/sampling_logp_difference/mean": 0.016956325620412827, |
| "step": 67, |
| "step_time": 16.57412173389457 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008019278873689472, |
| "clip_ratio/high_mean": 0.0008019278873689472, |
| "clip_ratio/low_mean": 7.771958771627397e-05, |
| "clip_ratio/low_min": 7.771958771627397e-05, |
| "clip_ratio/region_mean": 0.0008796474896371365, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1519.0, |
| "completions/max_terminated_length": 1519.0, |
| "completions/mean_length": 728.5, |
| "completions/mean_terminated_length": 728.5, |
| "completions/min_length": 434.0, |
| "completions/min_terminated_length": 434.0, |
| "entropy": 0.1393839403986931, |
| "epoch": 0.18478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01587974652647972, |
| "learning_rate": 5e-05, |
| "loss": 0.029773788526654243, |
| "num_tokens": 4544992.0, |
| "reward": 0.44428709149360657, |
| "reward_std": 0.5228669047355652, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.355712890625, |
| "rewards/length_penalty/std": 0.12850673496723175, |
| "sampling/importance_sampling_ratio/max": 2.582442045211792, |
| "sampling/importance_sampling_ratio/mean": 0.9948198795318604, |
| "sampling/importance_sampling_ratio/min": 0.14034956693649292, |
| "sampling/sampling_logp_difference/max": 1.9636191129684448, |
| "sampling/sampling_logp_difference/mean": 0.017770489677786827, |
| "step": 68, |
| "step_time": 17.422449697973207 |
| }, |
| { |
| "clip_ratio/high_max": 0.000864779035327956, |
| "clip_ratio/high_mean": 0.000864779035327956, |
| "clip_ratio/low_mean": 0.00015088155632838606, |
| "clip_ratio/low_min": 0.00015088155632838606, |
| "clip_ratio/region_mean": 0.0010156606091186403, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1994.0, |
| "completions/mean_length": 702.5799560546875, |
| "completions/mean_terminated_length": 585.5869750976562, |
| "completions/min_length": 326.0, |
| "completions/min_terminated_length": 326.0, |
| "entropy": 0.22556258738040924, |
| "epoch": 0.1875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016324201598763466, |
| "learning_rate": 5e-05, |
| "loss": 0.05751039832830429, |
| "num_tokens": 4582501.0, |
| "reward": 0.25694334506988525, |
| "reward_std": 0.6827882528305054, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.34305664896965027, |
| "rewards/length_penalty/std": 0.26258355379104614, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9914600253105164, |
| "sampling/importance_sampling_ratio/min": 0.15051023662090302, |
| "sampling/sampling_logp_difference/max": 1.8937242031097412, |
| "sampling/sampling_logp_difference/mean": 0.023742323741316795, |
| "step": 69, |
| "step_time": 22.759062204044312 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006425084196962416, |
| "clip_ratio/high_mean": 0.0006425084196962416, |
| "clip_ratio/low_mean": 8.321676723426208e-05, |
| "clip_ratio/low_min": 8.321676723426208e-05, |
| "clip_ratio/region_mean": 0.0007257252116687595, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1504.0, |
| "completions/mean_length": 955.2799682617188, |
| "completions/mean_terminated_length": 647.076904296875, |
| "completions/min_length": 285.0, |
| "completions/min_terminated_length": 285.0, |
| "entropy": 0.13123125433921815, |
| "epoch": 0.19021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015733294188976288, |
| "learning_rate": 5e-05, |
| "loss": 0.05914657935500145, |
| "num_tokens": 4633775.0, |
| "reward": -0.2664453089237213, |
| "reward_std": 0.5910485982894897, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.46644532680511475, |
| "rewards/length_penalty/std": 0.31597575545310974, |
| "sampling/importance_sampling_ratio/max": 2.3739559650421143, |
| "sampling/importance_sampling_ratio/mean": 0.9944911599159241, |
| "sampling/importance_sampling_ratio/min": 0.10365132242441177, |
| "sampling/sampling_logp_difference/max": 2.2667226791381836, |
| "sampling/sampling_logp_difference/mean": 0.01540339831262827, |
| "step": 70, |
| "step_time": 24.37746852566488 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007989955949597061, |
| "clip_ratio/high_mean": 0.0007989955949597061, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007989955949597061, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 736.0, |
| "completions/mean_length": 498.3599853515625, |
| "completions/mean_terminated_length": 466.73468017578125, |
| "completions/min_length": 277.0, |
| "completions/min_terminated_length": 277.0, |
| "entropy": 0.1765614241361618, |
| "epoch": 0.19293478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015742355957627296, |
| "learning_rate": 5e-05, |
| "loss": 0.056793928146362305, |
| "num_tokens": 4661053.0, |
| "reward": 0.2966601550579071, |
| "reward_std": 0.5330518484115601, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.24333983659744263, |
| "rewards/length_penalty/std": 0.12934282422065735, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9933785200119019, |
| "sampling/importance_sampling_ratio/min": 0.07459692656993866, |
| "sampling/sampling_logp_difference/max": 2.595655918121338, |
| "sampling/sampling_logp_difference/mean": 0.024273915216326714, |
| "step": 71, |
| "step_time": 21.010152561357245 |
| }, |
| { |
| "clip_ratio/high_max": 0.00042127874912694095, |
| "clip_ratio/high_mean": 0.00042127874912694095, |
| "clip_ratio/low_mean": 3.858769196085632e-05, |
| "clip_ratio/low_min": 3.858769196085632e-05, |
| "clip_ratio/region_mean": 0.0004598664410877973, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 896.0, |
| "completions/max_terminated_length": 896.0, |
| "completions/mean_length": 522.8399658203125, |
| "completions/mean_terminated_length": 522.8399658203125, |
| "completions/min_length": 281.0, |
| "completions/min_terminated_length": 281.0, |
| "entropy": 0.13467985838651658, |
| "epoch": 0.1956521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016225093975663185, |
| "learning_rate": 5e-05, |
| "loss": 0.022949673235416412, |
| "num_tokens": 4690865.0, |
| "reward": 0.4847070276737213, |
| "reward_std": 0.5070469975471497, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.25529298186302185, |
| "rewards/length_penalty/std": 0.08098991215229034, |
| "sampling/importance_sampling_ratio/max": 2.6677114963531494, |
| "sampling/importance_sampling_ratio/mean": 0.9940216541290283, |
| "sampling/importance_sampling_ratio/min": 0.17645502090454102, |
| "sampling/sampling_logp_difference/max": 1.7346893548965454, |
| "sampling/sampling_logp_difference/mean": 0.0198117196559906, |
| "step": 72, |
| "step_time": 10.840452854987234 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007927220955025405, |
| "clip_ratio/high_mean": 0.0007927220955025405, |
| "clip_ratio/low_mean": 9.946971840690821e-05, |
| "clip_ratio/low_min": 9.946971840690821e-05, |
| "clip_ratio/region_mean": 0.0008921918226405979, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1651.0, |
| "completions/mean_length": 780.3399658203125, |
| "completions/mean_terminated_length": 607.477294921875, |
| "completions/min_length": 251.0, |
| "completions/min_terminated_length": 251.0, |
| "entropy": 0.23456223607063292, |
| "epoch": 0.1983695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020184226334095, |
| "learning_rate": 5e-05, |
| "loss": 0.05907906964421272, |
| "num_tokens": 4732292.0, |
| "reward": 0.23897460103034973, |
| "reward_std": 0.6833838820457458, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.38102540373802185, |
| "rewards/length_penalty/std": 0.2713907063007355, |
| "sampling/importance_sampling_ratio/max": 2.268763780593872, |
| "sampling/importance_sampling_ratio/mean": 0.9913050532341003, |
| "sampling/importance_sampling_ratio/min": 0.2665638029575348, |
| "sampling/sampling_logp_difference/max": 1.3221416473388672, |
| "sampling/sampling_logp_difference/mean": 0.023112887516617775, |
| "step": 73, |
| "step_time": 23.074215144850314 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006163164565805346, |
| "clip_ratio/high_mean": 0.0006163164565805346, |
| "clip_ratio/low_mean": 0.0001756874320562929, |
| "clip_ratio/low_min": 0.0001756874320562929, |
| "clip_ratio/region_mean": 0.0007920038769952953, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1442.0, |
| "completions/max_terminated_length": 1442.0, |
| "completions/mean_length": 553.1799926757812, |
| "completions/mean_terminated_length": 553.1799926757812, |
| "completions/min_length": 263.0, |
| "completions/min_terminated_length": 263.0, |
| "entropy": 0.18717554807662964, |
| "epoch": 0.20108695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018485983833670616, |
| "learning_rate": 5e-05, |
| "loss": 0.06435728073120117, |
| "num_tokens": 4763281.0, |
| "reward": 0.12989257276058197, |
| "reward_std": 0.5527178049087524, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.2701074182987213, |
| "rewards/length_penalty/std": 0.12861602008342743, |
| "sampling/importance_sampling_ratio/max": 2.640171527862549, |
| "sampling/importance_sampling_ratio/mean": 0.9927074909210205, |
| "sampling/importance_sampling_ratio/min": 0.23624838888645172, |
| "sampling/sampling_logp_difference/max": 1.4428715705871582, |
| "sampling/sampling_logp_difference/mean": 0.022419407963752747, |
| "step": 74, |
| "step_time": 16.578645288711414 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009462563553825021, |
| "clip_ratio/high_mean": 0.0009462563553825021, |
| "clip_ratio/low_mean": 0.0001240236306330189, |
| "clip_ratio/low_min": 0.0001240236306330189, |
| "clip_ratio/region_mean": 0.001070279988925904, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1158.0, |
| "completions/max_terminated_length": 1158.0, |
| "completions/mean_length": 461.8199768066406, |
| "completions/mean_terminated_length": 461.8199768066406, |
| "completions/min_length": 151.0, |
| "completions/min_terminated_length": 151.0, |
| "entropy": 0.14905855804681778, |
| "epoch": 0.20380434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01307652611285448, |
| "learning_rate": 5e-05, |
| "loss": 0.021004769951105118, |
| "num_tokens": 4788702.0, |
| "reward": 0.6745019555091858, |
| "reward_std": 0.3246726393699646, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304577946662903, |
| "rewards/length_penalty/mean": -0.2254980504512787, |
| "rewards/length_penalty/std": 0.14760982990264893, |
| "sampling/importance_sampling_ratio/max": 2.4556875228881836, |
| "sampling/importance_sampling_ratio/mean": 0.9943245649337769, |
| "sampling/importance_sampling_ratio/min": 0.10875154286623001, |
| "sampling/sampling_logp_difference/max": 2.218689441680908, |
| "sampling/sampling_logp_difference/mean": 0.021682288497686386, |
| "step": 75, |
| "step_time": 12.870562972733751 |
| }, |
| { |
| "clip_ratio/high_max": 0.00037010837986599653, |
| "clip_ratio/high_mean": 0.00037010837986599653, |
| "clip_ratio/low_mean": 0.00023711871763225644, |
| "clip_ratio/low_min": 0.00023711871763225644, |
| "clip_ratio/region_mean": 0.0006072270858567208, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2034.0, |
| "completions/mean_length": 580.260009765625, |
| "completions/mean_terminated_length": 486.574462890625, |
| "completions/min_length": 157.0, |
| "completions/min_terminated_length": 157.0, |
| "entropy": 0.22566772997379303, |
| "epoch": 0.20652173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016523700207471848, |
| "learning_rate": 5e-05, |
| "loss": 0.05377320945262909, |
| "num_tokens": 4821055.0, |
| "reward": 0.4966699182987213, |
| "reward_std": 0.6526171565055847, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.2833300828933716, |
| "rewards/length_penalty/std": 0.2616604268550873, |
| "sampling/importance_sampling_ratio/max": 2.3889684677124023, |
| "sampling/importance_sampling_ratio/mean": 0.9903892278671265, |
| "sampling/importance_sampling_ratio/min": 0.17624707520008087, |
| "sampling/sampling_logp_difference/max": 1.7358684539794922, |
| "sampling/sampling_logp_difference/mean": 0.025178341194987297, |
| "step": 76, |
| "step_time": 22.302102555288002 |
| }, |
| { |
| "clip_ratio/high_max": 0.000773135747294873, |
| "clip_ratio/high_mean": 0.000773135747294873, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000773135747294873, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1668.0, |
| "completions/mean_length": 512.4599609375, |
| "completions/mean_terminated_length": 481.1224365234375, |
| "completions/min_length": 122.0, |
| "completions/min_terminated_length": 122.0, |
| "entropy": 0.21672658920288085, |
| "epoch": 0.20923913043478262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015378485433757305, |
| "learning_rate": 5e-05, |
| "loss": 0.043909382075071335, |
| "num_tokens": 4849238.0, |
| "reward": 0.5897753834724426, |
| "reward_std": 0.5213438272476196, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.25022462010383606, |
| "rewards/length_penalty/std": 0.2067212611436844, |
| "sampling/importance_sampling_ratio/max": 2.3792941570281982, |
| "sampling/importance_sampling_ratio/mean": 0.9912435412406921, |
| "sampling/importance_sampling_ratio/min": 0.19379664957523346, |
| "sampling/sampling_logp_difference/max": 1.6409459114074707, |
| "sampling/sampling_logp_difference/mean": 0.02349938079714775, |
| "step": 77, |
| "step_time": 21.625199718866497 |
| }, |
| { |
| "clip_ratio/high_max": 0.000660551741020754, |
| "clip_ratio/high_mean": 0.000660551741020754, |
| "clip_ratio/low_mean": 6.988528766669334e-05, |
| "clip_ratio/low_min": 6.988528766669334e-05, |
| "clip_ratio/region_mean": 0.0007304370228666812, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1316.0, |
| "completions/max_terminated_length": 1316.0, |
| "completions/mean_length": 496.8999938964844, |
| "completions/mean_terminated_length": 496.8999938964844, |
| "completions/min_length": 233.0, |
| "completions/min_terminated_length": 233.0, |
| "entropy": 0.10440312922000886, |
| "epoch": 0.21195652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012265080586075783, |
| "learning_rate": 5e-05, |
| "loss": 0.04140622168779373, |
| "num_tokens": 4876383.0, |
| "reward": 0.35737302899360657, |
| "reward_std": 0.584845781326294, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.24262695014476776, |
| "rewards/length_penalty/std": 0.1352190375328064, |
| "sampling/importance_sampling_ratio/max": 2.9317548274993896, |
| "sampling/importance_sampling_ratio/mean": 0.995967447757721, |
| "sampling/importance_sampling_ratio/min": 0.10318702459335327, |
| "sampling/sampling_logp_difference/max": 2.271212100982666, |
| "sampling/sampling_logp_difference/mean": 0.016470689326524734, |
| "step": 78, |
| "step_time": 14.913867953931913 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005591347988229245, |
| "clip_ratio/high_mean": 0.0005591347988229245, |
| "clip_ratio/low_mean": 0.00012265780533198266, |
| "clip_ratio/low_min": 0.00012265780533198266, |
| "clip_ratio/region_mean": 0.0006817926187068224, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 791.0, |
| "completions/max_terminated_length": 791.0, |
| "completions/mean_length": 508.8199768066406, |
| "completions/mean_terminated_length": 508.8199768066406, |
| "completions/min_length": 300.0, |
| "completions/min_terminated_length": 300.0, |
| "entropy": 0.13780646920204162, |
| "epoch": 0.21467391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017290102317929268, |
| "learning_rate": 5e-05, |
| "loss": 0.031766217201948166, |
| "num_tokens": 4905284.0, |
| "reward": 0.5315527319908142, |
| "reward_std": 0.4167904555797577, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.2484472692012787, |
| "rewards/length_penalty/std": 0.0692163035273552, |
| "sampling/importance_sampling_ratio/max": 2.2687551975250244, |
| "sampling/importance_sampling_ratio/mean": 0.9940080046653748, |
| "sampling/importance_sampling_ratio/min": 0.19560691714286804, |
| "sampling/sampling_logp_difference/max": 1.631648063659668, |
| "sampling/sampling_logp_difference/mean": 0.022410748526453972, |
| "step": 79, |
| "step_time": 9.642815279541537 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005169936892343685, |
| "clip_ratio/high_mean": 0.0005169936892343685, |
| "clip_ratio/low_mean": 0.0001329172489931807, |
| "clip_ratio/low_min": 0.0001329172489931807, |
| "clip_ratio/region_mean": 0.0006499109382275492, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 745.0, |
| "completions/max_terminated_length": 745.0, |
| "completions/mean_length": 434.47998046875, |
| "completions/mean_terminated_length": 434.47998046875, |
| "completions/min_length": 186.0, |
| "completions/min_terminated_length": 186.0, |
| "entropy": 0.16563438773155212, |
| "epoch": 0.21739130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018692871555685997, |
| "learning_rate": 5e-05, |
| "loss": 0.03147809952497482, |
| "num_tokens": 4930168.0, |
| "reward": 0.18785156309604645, |
| "reward_std": 0.5391013622283936, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.21214844286441803, |
| "rewards/length_penalty/std": 0.06361670047044754, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9938739538192749, |
| "sampling/importance_sampling_ratio/min": 0.1721082478761673, |
| "sampling/sampling_logp_difference/max": 1.759631633758545, |
| "sampling/sampling_logp_difference/mean": 0.02338912896811962, |
| "step": 80, |
| "step_time": 8.96043543703854 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006292079895501956, |
| "clip_ratio/high_mean": 0.0006292079895501956, |
| "clip_ratio/low_mean": 9.467342169955373e-05, |
| "clip_ratio/low_min": 9.467342169955373e-05, |
| "clip_ratio/region_mean": 0.0007238814112497493, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 904.0, |
| "completions/max_terminated_length": 904.0, |
| "completions/mean_length": 456.0799865722656, |
| "completions/mean_terminated_length": 456.0799865722656, |
| "completions/min_length": 215.0, |
| "completions/min_terminated_length": 215.0, |
| "entropy": 0.14513773322105408, |
| "epoch": 0.22010869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016790563240647316, |
| "learning_rate": 5e-05, |
| "loss": 0.03664315864443779, |
| "num_tokens": 4957432.0, |
| "reward": 0.537304699420929, |
| "reward_std": 0.39452749490737915, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.22269530594348907, |
| "rewards/length_penalty/std": 0.07295688986778259, |
| "sampling/importance_sampling_ratio/max": 2.5686287879943848, |
| "sampling/importance_sampling_ratio/mean": 0.9947906136512756, |
| "sampling/importance_sampling_ratio/min": 0.3118853271007538, |
| "sampling/sampling_logp_difference/max": 1.1651196479797363, |
| "sampling/sampling_logp_difference/mean": 0.02233206480741501, |
| "step": 81, |
| "step_time": 10.900564274052158 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007119274698197842, |
| "clip_ratio/high_mean": 0.0007119274698197842, |
| "clip_ratio/low_mean": 6.240249495021999e-05, |
| "clip_ratio/low_min": 6.240249495021999e-05, |
| "clip_ratio/region_mean": 0.0007743299705907702, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 592.0, |
| "completions/max_terminated_length": 592.0, |
| "completions/mean_length": 311.6999816894531, |
| "completions/mean_terminated_length": 311.6999816894531, |
| "completions/min_length": 145.0, |
| "completions/min_terminated_length": 145.0, |
| "entropy": 0.1522587776184082, |
| "epoch": 0.22282608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012664209119975567, |
| "learning_rate": 5e-05, |
| "loss": 0.01738598197698593, |
| "num_tokens": 4975867.0, |
| "reward": 0.5078027248382568, |
| "reward_std": 0.5126169919967651, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.15219727158546448, |
| "rewards/length_penalty/std": 0.05498060956597328, |
| "sampling/importance_sampling_ratio/max": 2.9442734718322754, |
| "sampling/importance_sampling_ratio/mean": 0.9939560294151306, |
| "sampling/importance_sampling_ratio/min": 0.0471280999481678, |
| "sampling/sampling_logp_difference/max": 3.0548858642578125, |
| "sampling/sampling_logp_difference/mean": 0.02900662086904049, |
| "step": 82, |
| "step_time": 7.0824445898178965 |
| }, |
| { |
| "clip_ratio/high_max": 0.00029795404989272354, |
| "clip_ratio/high_mean": 0.00029795404989272354, |
| "clip_ratio/low_mean": 0.00024681565118953583, |
| "clip_ratio/low_min": 0.00024681565118953583, |
| "clip_ratio/region_mean": 0.0005447697127237916, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 484.0, |
| "completions/max_terminated_length": 484.0, |
| "completions/mean_length": 250.75999450683594, |
| "completions/mean_terminated_length": 250.75999450683594, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.14524978399276733, |
| "epoch": 0.22554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01768699288368225, |
| "learning_rate": 5e-05, |
| "loss": 0.019604604691267014, |
| "num_tokens": 4990835.0, |
| "reward": 0.4575585722923279, |
| "reward_std": 0.5098227858543396, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.12244140356779099, |
| "rewards/length_penalty/std": 0.04124342277646065, |
| "sampling/importance_sampling_ratio/max": 2.6671459674835205, |
| "sampling/importance_sampling_ratio/mean": 0.9945991635322571, |
| "sampling/importance_sampling_ratio/min": 0.27584001421928406, |
| "sampling/sampling_logp_difference/max": 1.2879343032836914, |
| "sampling/sampling_logp_difference/mean": 0.024842264130711555, |
| "step": 83, |
| "step_time": 5.764580278657377 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005242552317213268, |
| "clip_ratio/high_mean": 0.0005242552317213268, |
| "clip_ratio/low_mean": 0.00016014102438930423, |
| "clip_ratio/low_min": 0.00016014102438930423, |
| "clip_ratio/region_mean": 0.0006843962473794818, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1553.0, |
| "completions/max_terminated_length": 1553.0, |
| "completions/mean_length": 410.94000244140625, |
| "completions/mean_terminated_length": 410.94000244140625, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.15034571588039397, |
| "epoch": 0.22826086956521738, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01646677404642105, |
| "learning_rate": 5e-05, |
| "loss": 0.03158365562558174, |
| "num_tokens": 5014362.0, |
| "reward": 0.5593456625938416, |
| "reward_std": 0.4266625940799713, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.2006542980670929, |
| "rewards/length_penalty/std": 0.18420180678367615, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9939844608306885, |
| "sampling/importance_sampling_ratio/min": 0.27052757143974304, |
| "sampling/sampling_logp_difference/max": 1.6090869903564453, |
| "sampling/sampling_logp_difference/mean": 0.023058636114001274, |
| "step": 84, |
| "step_time": 17.104518838925287 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012515002861618995, |
| "clip_ratio/high_mean": 0.0012515002861618995, |
| "clip_ratio/low_mean": 0.00010666666785255074, |
| "clip_ratio/low_min": 0.00010666666785255074, |
| "clip_ratio/region_mean": 0.0013581669540144504, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1174.0, |
| "completions/max_terminated_length": 1174.0, |
| "completions/mean_length": 375.2599792480469, |
| "completions/mean_terminated_length": 375.2599792480469, |
| "completions/min_length": 139.0, |
| "completions/min_terminated_length": 139.0, |
| "entropy": 0.2758207768201828, |
| "epoch": 0.23097826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017457248643040657, |
| "learning_rate": 5e-05, |
| "loss": 0.033208977431058884, |
| "num_tokens": 5035145.0, |
| "reward": 0.436767578125, |
| "reward_std": 0.521892786026001, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.18323242664337158, |
| "rewards/length_penalty/std": 0.12026773393154144, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9894677400588989, |
| "sampling/importance_sampling_ratio/min": 0.2419632375240326, |
| "sampling/sampling_logp_difference/max": 1.4875578880310059, |
| "sampling/sampling_logp_difference/mean": 0.03322805464267731, |
| "step": 85, |
| "step_time": 12.625046013388783 |
| }, |
| { |
| "clip_ratio/high_max": 0.001223215099889785, |
| "clip_ratio/high_mean": 0.001223215099889785, |
| "clip_ratio/low_mean": 0.00030505707254633305, |
| "clip_ratio/low_min": 0.00030505707254633305, |
| "clip_ratio/region_mean": 0.0015282721840776503, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 319.0, |
| "completions/max_terminated_length": 319.0, |
| "completions/mean_length": 187.3199920654297, |
| "completions/mean_terminated_length": 187.3199920654297, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.16413030922412872, |
| "epoch": 0.23369565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011740758083760738, |
| "learning_rate": 5e-05, |
| "loss": 0.012568159028887749, |
| "num_tokens": 5046611.0, |
| "reward": 0.8285351395606995, |
| "reward_std": 0.27608367800712585, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.09146484732627869, |
| "rewards/length_penalty/std": 0.02840513363480568, |
| "sampling/importance_sampling_ratio/max": 2.6626501083374023, |
| "sampling/importance_sampling_ratio/mean": 0.9936551451683044, |
| "sampling/importance_sampling_ratio/min": 0.173045352101326, |
| "sampling/sampling_logp_difference/max": 1.7542015314102173, |
| "sampling/sampling_logp_difference/mean": 0.03142157942056656, |
| "step": 86, |
| "step_time": 4.042918574763462 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007650235609617084, |
| "clip_ratio/high_mean": 0.0007650235609617084, |
| "clip_ratio/low_mean": 0.0002287703042384237, |
| "clip_ratio/low_min": 0.0002287703042384237, |
| "clip_ratio/region_mean": 0.0009937938652001322, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 702.0, |
| "completions/max_terminated_length": 702.0, |
| "completions/mean_length": 275.8800048828125, |
| "completions/mean_terminated_length": 275.8800048828125, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.16404963433742523, |
| "epoch": 0.23641304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016037670895457268, |
| "learning_rate": 5e-05, |
| "loss": 0.030306950211524963, |
| "num_tokens": 5062615.0, |
| "reward": 0.42529296875, |
| "reward_std": 0.5270707607269287, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.1347070336341858, |
| "rewards/length_penalty/std": 0.06707599014043808, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9937471151351929, |
| "sampling/importance_sampling_ratio/min": 0.2099948674440384, |
| "sampling/sampling_logp_difference/max": 1.5606721639633179, |
| "sampling/sampling_logp_difference/mean": 0.0274569820612669, |
| "step": 87, |
| "step_time": 7.923893882660195 |
| }, |
| { |
| "clip_ratio/high_max": 0.001012345578055829, |
| "clip_ratio/high_mean": 0.001012345578055829, |
| "clip_ratio/low_mean": 0.000374765763990581, |
| "clip_ratio/low_min": 0.000374765763990581, |
| "clip_ratio/region_mean": 0.0013871113187633455, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 450.0, |
| "completions/max_terminated_length": 450.0, |
| "completions/mean_length": 183.95999145507812, |
| "completions/mean_terminated_length": 183.95999145507812, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.18258396983146669, |
| "epoch": 0.2391304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013715540058910847, |
| "learning_rate": 5e-05, |
| "loss": 0.014025804586708546, |
| "num_tokens": 5074703.0, |
| "reward": 0.8101757764816284, |
| "reward_std": 0.3099156320095062, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304574966430664, |
| "rewards/length_penalty/mean": -0.08982422202825546, |
| "rewards/length_penalty/std": 0.04726439714431763, |
| "sampling/importance_sampling_ratio/max": 2.758549451828003, |
| "sampling/importance_sampling_ratio/mean": 0.993539035320282, |
| "sampling/importance_sampling_ratio/min": 0.22761517763137817, |
| "sampling/sampling_logp_difference/max": 1.4800989627838135, |
| "sampling/sampling_logp_difference/mean": 0.02929404191672802, |
| "step": 88, |
| "step_time": 5.350178437074646 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008073070493992418, |
| "clip_ratio/high_mean": 0.0008073070493992418, |
| "clip_ratio/low_mean": 8.908685995265842e-05, |
| "clip_ratio/low_min": 8.908685995265842e-05, |
| "clip_ratio/region_mean": 0.0008963939093519002, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 769.0, |
| "completions/max_terminated_length": 769.0, |
| "completions/mean_length": 228.95999145507812, |
| "completions/mean_terminated_length": 228.95999145507812, |
| "completions/min_length": 64.0, |
| "completions/min_terminated_length": 64.0, |
| "entropy": 0.21333446800708772, |
| "epoch": 0.2418478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014769138768315315, |
| "learning_rate": 5e-05, |
| "loss": 0.025534415617585182, |
| "num_tokens": 5089611.0, |
| "reward": 0.2482031136751175, |
| "reward_std": 0.5174915194511414, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.11179687827825546, |
| "rewards/length_penalty/std": 0.07019683718681335, |
| "sampling/importance_sampling_ratio/max": 2.8253140449523926, |
| "sampling/importance_sampling_ratio/mean": 0.9898026585578918, |
| "sampling/importance_sampling_ratio/min": 0.14434055984020233, |
| "sampling/sampling_logp_difference/max": 1.935579776763916, |
| "sampling/sampling_logp_difference/mean": 0.03491024300456047, |
| "step": 89, |
| "step_time": 8.411637461045757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010897691186983137, |
| "clip_ratio/high_mean": 0.0010897691186983137, |
| "clip_ratio/low_mean": 0.0003799433005042374, |
| "clip_ratio/low_min": 0.0003799433005042374, |
| "clip_ratio/region_mean": 0.0014697124192025513, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1294.0, |
| "completions/max_terminated_length": 1294.0, |
| "completions/mean_length": 320.239990234375, |
| "completions/mean_terminated_length": 320.239990234375, |
| "completions/min_length": 101.0, |
| "completions/min_terminated_length": 101.0, |
| "entropy": 0.20011376440525055, |
| "epoch": 0.24456521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013035651296377182, |
| "learning_rate": 5e-05, |
| "loss": 0.01909465715289116, |
| "num_tokens": 5108653.0, |
| "reward": 0.523632824420929, |
| "reward_std": 0.49329641461372375, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.47121208906173706, |
| "rewards/length_penalty/mean": -0.15636718273162842, |
| "rewards/length_penalty/std": 0.14421890676021576, |
| "sampling/importance_sampling_ratio/max": 2.7216994762420654, |
| "sampling/importance_sampling_ratio/mean": 0.9917253255844116, |
| "sampling/importance_sampling_ratio/min": 0.16296742856502533, |
| "sampling/sampling_logp_difference/max": 1.8142049312591553, |
| "sampling/sampling_logp_difference/mean": 0.02891269139945507, |
| "step": 90, |
| "step_time": 13.601352974073961 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007387871155515313, |
| "clip_ratio/high_mean": 0.0007387871155515313, |
| "clip_ratio/low_mean": 0.0009060284704901278, |
| "clip_ratio/low_min": 0.0009060284704901278, |
| "clip_ratio/region_mean": 0.0016448155976831913, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 801.0, |
| "completions/max_terminated_length": 801.0, |
| "completions/mean_length": 183.83999633789062, |
| "completions/mean_terminated_length": 183.83999633789062, |
| "completions/min_length": 40.0, |
| "completions/min_terminated_length": 40.0, |
| "entropy": 0.20541379153728484, |
| "epoch": 0.24728260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014283214695751667, |
| "learning_rate": 5e-05, |
| "loss": 0.01677056774497032, |
| "num_tokens": 5123015.0, |
| "reward": 0.690234363079071, |
| "reward_std": 0.43638521432876587, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.08976562321186066, |
| "rewards/length_penalty/std": 0.0880991667509079, |
| "sampling/importance_sampling_ratio/max": 2.830004930496216, |
| "sampling/importance_sampling_ratio/mean": 0.9925782084465027, |
| "sampling/importance_sampling_ratio/min": 0.10747049748897552, |
| "sampling/sampling_logp_difference/max": 2.230538845062256, |
| "sampling/sampling_logp_difference/mean": 0.030657807365059853, |
| "step": 91, |
| "step_time": 9.6455499723088 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013442946132272483, |
| "clip_ratio/high_mean": 0.0013442946132272483, |
| "clip_ratio/low_mean": 0.0010825463919900357, |
| "clip_ratio/low_min": 0.0010825463919900357, |
| "clip_ratio/region_mean": 0.0024268409935757516, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 283.0, |
| "completions/max_terminated_length": 283.0, |
| "completions/mean_length": 139.02000427246094, |
| "completions/mean_terminated_length": 139.02000427246094, |
| "completions/min_length": 59.0, |
| "completions/min_terminated_length": 59.0, |
| "entropy": 0.1973792403936386, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01335429772734642, |
| "learning_rate": 5e-05, |
| "loss": 0.013862542808055878, |
| "num_tokens": 5132356.0, |
| "reward": 0.5321191549301147, |
| "reward_std": 0.5006313920021057, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.06788086146116257, |
| "rewards/length_penalty/std": 0.026449084281921387, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9915658235549927, |
| "sampling/importance_sampling_ratio/min": 0.13321897387504578, |
| "sampling/sampling_logp_difference/max": 2.015761137008667, |
| "sampling/sampling_logp_difference/mean": 0.036115266382694244, |
| "step": 92, |
| "step_time": 3.6328126138541847 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016165593988262117, |
| "clip_ratio/high_mean": 0.0016165593988262117, |
| "clip_ratio/low_mean": 0.0006566167459823192, |
| "clip_ratio/low_min": 0.0006566167459823192, |
| "clip_ratio/region_mean": 0.0022731761331669987, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 514.0, |
| "completions/max_terminated_length": 514.0, |
| "completions/mean_length": 123.5999984741211, |
| "completions/mean_terminated_length": 123.5999984741211, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.31247851252555847, |
| "epoch": 0.25271739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01579795591533184, |
| "learning_rate": 5e-05, |
| "loss": 0.015175119042396545, |
| "num_tokens": 5144156.0, |
| "reward": 0.43964841961860657, |
| "reward_std": 0.5186566710472107, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.06035156175494194, |
| "rewards/length_penalty/std": 0.03829424828290939, |
| "sampling/importance_sampling_ratio/max": 2.9564101696014404, |
| "sampling/importance_sampling_ratio/mean": 0.9889784455299377, |
| "sampling/importance_sampling_ratio/min": 0.06740635633468628, |
| "sampling/sampling_logp_difference/max": 2.6970160007476807, |
| "sampling/sampling_logp_difference/mean": 0.05142742395401001, |
| "step": 93, |
| "step_time": 6.132455758051947 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009128096513450146, |
| "clip_ratio/high_mean": 0.0009128096513450146, |
| "clip_ratio/low_mean": 0.000355527934152633, |
| "clip_ratio/low_min": 0.000355527934152633, |
| "clip_ratio/region_mean": 0.0012683375854976476, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 448.0, |
| "completions/max_terminated_length": 448.0, |
| "completions/mean_length": 162.47999572753906, |
| "completions/mean_terminated_length": 162.47999572753906, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.2579308182001114, |
| "epoch": 0.2554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014927428215742111, |
| "learning_rate": 5e-05, |
| "loss": 0.014601172879338264, |
| "num_tokens": 5158310.0, |
| "reward": 0.3406640589237213, |
| "reward_std": 0.47975119948387146, |
| "rewards/correctness/mean": 0.41999998688697815, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.07933593541383743, |
| "rewards/length_penalty/std": 0.05228535458445549, |
| "sampling/importance_sampling_ratio/max": 2.5284106731414795, |
| "sampling/importance_sampling_ratio/mean": 0.9883469939231873, |
| "sampling/importance_sampling_ratio/min": 0.034295160323381424, |
| "sampling/sampling_logp_difference/max": 3.372750997543335, |
| "sampling/sampling_logp_difference/mean": 0.03882637247443199, |
| "step": 94, |
| "step_time": 6.023596244864166 |
| }, |
| { |
| "clip_ratio/high_max": 0.0022477660793811085, |
| "clip_ratio/high_mean": 0.0022477660793811085, |
| "clip_ratio/low_mean": 0.0012457157718017697, |
| "clip_ratio/low_min": 0.0012457157718017697, |
| "clip_ratio/region_mean": 0.0034934818977490067, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 144.0, |
| "completions/max_terminated_length": 144.0, |
| "completions/mean_length": 78.77999877929688, |
| "completions/mean_terminated_length": 78.77999877929688, |
| "completions/min_length": 31.0, |
| "completions/min_terminated_length": 31.0, |
| "entropy": 0.30045082569122317, |
| "epoch": 0.25815217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016335880383849144, |
| "learning_rate": 5e-05, |
| "loss": 0.006531589664518833, |
| "num_tokens": 5164619.0, |
| "reward": 0.321533203125, |
| "reward_std": 0.48154181241989136, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.03846679627895355, |
| "rewards/length_penalty/std": 0.015084164217114449, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.989437460899353, |
| "sampling/importance_sampling_ratio/min": 0.12465248256921768, |
| "sampling/sampling_logp_difference/max": 2.0822255611419678, |
| "sampling/sampling_logp_difference/mean": 0.05184780806303024, |
| "step": 95, |
| "step_time": 2.3817885289900005 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002816901309415698, |
| "clip_ratio/high_mean": 0.0002816901309415698, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0002816901309415698, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 144.0, |
| "completions/max_terminated_length": 144.0, |
| "completions/mean_length": 66.9000015258789, |
| "completions/mean_terminated_length": 66.9000015258789, |
| "completions/min_length": 28.0, |
| "completions/min_terminated_length": 28.0, |
| "entropy": 0.26627289950847627, |
| "epoch": 0.2608695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014192100614309311, |
| "learning_rate": 5e-05, |
| "loss": 0.0067526549100875854, |
| "num_tokens": 5171494.0, |
| "reward": 0.567333996295929, |
| "reward_std": 0.4992888867855072, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.03266601637005806, |
| "rewards/length_penalty/std": 0.013647029176354408, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9894942045211792, |
| "sampling/importance_sampling_ratio/min": 0.2218926101922989, |
| "sampling/sampling_logp_difference/max": 1.5055617094039917, |
| "sampling/sampling_logp_difference/mean": 0.04113667085766792, |
| "step": 96, |
| "step_time": 2.4275573221966624 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011642629513517022, |
| "clip_ratio/high_mean": 0.0011642629513517022, |
| "clip_ratio/low_mean": 0.0006319924490526318, |
| "clip_ratio/low_min": 0.0006319924490526318, |
| "clip_ratio/region_mean": 0.0017962554004043341, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 163.0, |
| "completions/max_terminated_length": 163.0, |
| "completions/mean_length": 68.43999481201172, |
| "completions/mean_terminated_length": 68.43999481201172, |
| "completions/min_length": 35.0, |
| "completions/min_terminated_length": 35.0, |
| "entropy": 0.266776242852211, |
| "epoch": 0.26358695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012949473224580288, |
| "learning_rate": 5e-05, |
| "loss": 0.0037423912435770035, |
| "num_tokens": 5177846.0, |
| "reward": 0.3665820360183716, |
| "reward_std": 0.49510350823402405, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.033417969942092896, |
| "rewards/length_penalty/std": 0.01156963873654604, |
| "sampling/importance_sampling_ratio/max": 2.1801586151123047, |
| "sampling/importance_sampling_ratio/mean": 0.9888060092926025, |
| "sampling/importance_sampling_ratio/min": 0.20605359971523285, |
| "sampling/sampling_logp_difference/max": 1.5796189308166504, |
| "sampling/sampling_logp_difference/mean": 0.040520939975976944, |
| "step": 97, |
| "step_time": 2.419445611303672 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 113.0, |
| "completions/max_terminated_length": 113.0, |
| "completions/mean_length": 47.779998779296875, |
| "completions/mean_terminated_length": 47.779998779296875, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.2810125112533569, |
| "epoch": 0.266304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013355037197470665, |
| "learning_rate": 5e-05, |
| "loss": 0.006037740036845207, |
| "num_tokens": 5182375.0, |
| "reward": 0.31666991114616394, |
| "reward_std": 0.4805067777633667, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.023330077528953552, |
| "rewards/length_penalty/std": 0.009400712326169014, |
| "sampling/importance_sampling_ratio/max": 2.7908573150634766, |
| "sampling/importance_sampling_ratio/mean": 0.9874352812767029, |
| "sampling/importance_sampling_ratio/min": 0.40323346853256226, |
| "sampling/sampling_logp_difference/max": 1.0263488292694092, |
| "sampling/sampling_logp_difference/mean": 0.042870886623859406, |
| "step": 98, |
| "step_time": 2.0863210430834442 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007498236140236258, |
| "clip_ratio/high_mean": 0.0007498236140236258, |
| "clip_ratio/low_mean": 0.0007629764266312122, |
| "clip_ratio/low_min": 0.0007629764266312122, |
| "clip_ratio/region_mean": 0.0015128000406548381, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 79.0, |
| "completions/max_terminated_length": 79.0, |
| "completions/mean_length": 49.81999969482422, |
| "completions/mean_terminated_length": 49.81999969482422, |
| "completions/min_length": 23.0, |
| "completions/min_terminated_length": 23.0, |
| "entropy": 0.23840693533420562, |
| "epoch": 0.26902173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009779835119843483, |
| "learning_rate": 5e-05, |
| "loss": 0.00247298926115036, |
| "num_tokens": 5187776.0, |
| "reward": 0.5956737995147705, |
| "reward_std": 0.4914281368255615, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.024326171725988388, |
| "rewards/length_penalty/std": 0.006979414261877537, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.989484965801239, |
| "sampling/importance_sampling_ratio/min": 0.21273456513881683, |
| "sampling/sampling_logp_difference/max": 1.5477100610733032, |
| "sampling/sampling_logp_difference/mean": 0.043070342391729355, |
| "step": 99, |
| "step_time": 1.8466221948619932 |
| }, |
| { |
| "clip_ratio/high_max": 0.001492611551657319, |
| "clip_ratio/high_mean": 0.001492611551657319, |
| "clip_ratio/low_mean": 0.0020396762527525427, |
| "clip_ratio/low_min": 0.0020396762527525427, |
| "clip_ratio/region_mean": 0.0035322878044098615, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 65.0, |
| "completions/max_terminated_length": 65.0, |
| "completions/mean_length": 39.13999938964844, |
| "completions/mean_terminated_length": 39.13999938964844, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.2374360203742981, |
| "epoch": 0.2717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011310559697449207, |
| "learning_rate": 5e-05, |
| "loss": 0.0034892666153609753, |
| "num_tokens": 5191963.0, |
| "reward": 0.6008886694908142, |
| "reward_std": 0.48954492807388306, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.019111327826976776, |
| "rewards/length_penalty/std": 0.006043337285518646, |
| "sampling/importance_sampling_ratio/max": 1.830500602722168, |
| "sampling/importance_sampling_ratio/mean": 0.9869400262832642, |
| "sampling/importance_sampling_ratio/min": 0.18310081958770752, |
| "sampling/sampling_logp_difference/max": 1.6977183818817139, |
| "sampling/sampling_logp_difference/mean": 0.04277850314974785, |
| "step": 100, |
| "step_time": 1.7810746191535145 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015503282193094492, |
| "clip_ratio/high_mean": 0.0015503282193094492, |
| "clip_ratio/low_mean": 0.000963855441659689, |
| "clip_ratio/low_min": 0.000963855441659689, |
| "clip_ratio/region_mean": 0.002514183660969138, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.0, |
| "completions/max_terminated_length": 71.0, |
| "completions/mean_length": 41.21999740600586, |
| "completions/mean_terminated_length": 41.21999740600586, |
| "completions/min_length": 21.0, |
| "completions/min_terminated_length": 21.0, |
| "entropy": 0.21979199945926667, |
| "epoch": 0.27445652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01346304640173912, |
| "learning_rate": 5e-05, |
| "loss": 0.0032673939131200314, |
| "num_tokens": 5196914.0, |
| "reward": 0.4398730397224426, |
| "reward_std": 0.503574013710022, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.020126953721046448, |
| "rewards/length_penalty/std": 0.007509734481573105, |
| "sampling/importance_sampling_ratio/max": 1.9471760988235474, |
| "sampling/importance_sampling_ratio/mean": 0.9929113984107971, |
| "sampling/importance_sampling_ratio/min": 0.43813568353652954, |
| "sampling/sampling_logp_difference/max": 0.8252266645431519, |
| "sampling/sampling_logp_difference/mean": 0.03735937178134918, |
| "step": 101, |
| "step_time": 1.8167982143349946 |
| }, |
| { |
| "clip_ratio/high_max": 0.0028410314582288263, |
| "clip_ratio/high_mean": 0.0028410314582288263, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0028410314582288263, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 74.0, |
| "completions/max_terminated_length": 74.0, |
| "completions/mean_length": 28.219999313354492, |
| "completions/mean_terminated_length": 28.219999313354492, |
| "completions/min_length": 17.0, |
| "completions/min_terminated_length": 17.0, |
| "entropy": 0.24149580895900727, |
| "epoch": 0.27717391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012711185961961746, |
| "learning_rate": 5e-05, |
| "loss": 0.001443374203518033, |
| "num_tokens": 5201255.0, |
| "reward": 0.4462206959724426, |
| "reward_std": 0.5012037754058838, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.013779296539723873, |
| "rewards/length_penalty/std": 0.005156445316970348, |
| "sampling/importance_sampling_ratio/max": 1.7634351253509521, |
| "sampling/importance_sampling_ratio/mean": 0.9892796874046326, |
| "sampling/importance_sampling_ratio/min": 0.1537872552871704, |
| "sampling/sampling_logp_difference/max": 1.8721851110458374, |
| "sampling/sampling_logp_difference/mean": 0.045835334807634354, |
| "step": 102, |
| "step_time": 2.262204034719616 |
| }, |
| { |
| "clip_ratio/high_max": 0.004400626383721828, |
| "clip_ratio/high_mean": 0.004400626383721828, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.004400626383721828, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 302.0, |
| "completions/max_terminated_length": 302.0, |
| "completions/mean_length": 46.37999725341797, |
| "completions/mean_terminated_length": 46.37999725341797, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.38878374695777895, |
| "epoch": 0.2798913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014794082380831242, |
| "learning_rate": 5e-05, |
| "loss": 0.01275131106376648, |
| "num_tokens": 5207814.0, |
| "reward": 0.17735351622104645, |
| "reward_std": 0.4058431386947632, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.022646484896540642, |
| "rewards/length_penalty/std": 0.02982097491621971, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9837145209312439, |
| "sampling/importance_sampling_ratio/min": 0.05253036320209503, |
| "sampling/sampling_logp_difference/max": 2.946363925933838, |
| "sampling/sampling_logp_difference/mean": 0.04632875323295593, |
| "step": 103, |
| "step_time": 3.9144057133235037 |
| }, |
| { |
| "clip_ratio/high_max": 0.001541568897664547, |
| "clip_ratio/high_mean": 0.001541568897664547, |
| "clip_ratio/low_mean": 0.0006535947788506747, |
| "clip_ratio/low_min": 0.0006535947788506747, |
| "clip_ratio/region_mean": 0.0021951636765152214, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 108.0, |
| "completions/max_terminated_length": 108.0, |
| "completions/mean_length": 25.53999900817871, |
| "completions/mean_terminated_length": 25.53999900817871, |
| "completions/min_length": 12.0, |
| "completions/min_terminated_length": 12.0, |
| "entropy": 0.29176886677742003, |
| "epoch": 0.2826086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013634953647851944, |
| "learning_rate": 5e-05, |
| "loss": 0.00450455117970705, |
| "num_tokens": 5212601.0, |
| "reward": 0.007529296912252903, |
| "reward_std": 0.14171697199344635, |
| "rewards/correctness/mean": 0.019999999552965164, |
| "rewards/correctness/std": 0.1414213627576828, |
| "rewards/length_penalty/mean": -0.012470703572034836, |
| "rewards/length_penalty/std": 0.007280715275555849, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.988826334476471, |
| "sampling/importance_sampling_ratio/min": 0.10926879197359085, |
| "sampling/sampling_logp_difference/max": 2.213944435119629, |
| "sampling/sampling_logp_difference/mean": 0.051908548921346664, |
| "step": 104, |
| "step_time": 1.9506963810417801 |
| }, |
| { |
| "clip_ratio/high_max": 0.005139961279928685, |
| "clip_ratio/high_mean": 0.005139961279928685, |
| "clip_ratio/low_mean": 0.0033204220235347748, |
| "clip_ratio/low_min": 0.0033204220235347748, |
| "clip_ratio/region_mean": 0.008460383210331202, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 45.0, |
| "completions/max_terminated_length": 45.0, |
| "completions/mean_length": 18.68000030517578, |
| "completions/mean_terminated_length": 18.68000030517578, |
| "completions/min_length": 12.0, |
| "completions/min_terminated_length": 12.0, |
| "entropy": 0.14449126720428468, |
| "epoch": 0.28532608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011553256772458553, |
| "learning_rate": 5e-05, |
| "loss": 0.001228717272169888, |
| "num_tokens": 5215895.0, |
| "reward": 0.2708789110183716, |
| "reward_std": 0.45316794514656067, |
| "rewards/correctness/mean": 0.2800000011920929, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.009121093899011612, |
| "rewards/length_penalty/std": 0.0039005656726658344, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9834380149841309, |
| "sampling/importance_sampling_ratio/min": 0.19683831930160522, |
| "sampling/sampling_logp_difference/max": 1.6253726482391357, |
| "sampling/sampling_logp_difference/mean": 0.05328484997153282, |
| "step": 105, |
| "step_time": 1.5497713307850063 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0013793103396892547, |
| "clip_ratio/low_min": 0.0013793103396892547, |
| "clip_ratio/region_mean": 0.0013793103396892547, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 24.0, |
| "completions/max_terminated_length": 24.0, |
| "completions/mean_length": 13.75999927520752, |
| "completions/mean_terminated_length": 13.75999927520752, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.15116987526416778, |
| "epoch": 0.28804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019801106303930283, |
| "learning_rate": 5e-05, |
| "loss": 0.0007544478867202997, |
| "num_tokens": 5219063.0, |
| "reward": 0.21328124403953552, |
| "reward_std": 0.4187287986278534, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.006718750111758709, |
| "rewards/length_penalty/std": 0.0023146714083850384, |
| "sampling/importance_sampling_ratio/max": 1.862955093383789, |
| "sampling/importance_sampling_ratio/mean": 0.9990727305412292, |
| "sampling/importance_sampling_ratio/min": 0.28229573369026184, |
| "sampling/sampling_logp_difference/max": 1.2648000717163086, |
| "sampling/sampling_logp_difference/mean": 0.03044041432440281, |
| "step": 106, |
| "step_time": 1.4075597082264721 |
| }, |
| { |
| "clip_ratio/high_max": 0.01647620350122452, |
| "clip_ratio/high_mean": 0.01647620350122452, |
| "clip_ratio/low_mean": 0.0035098522901535036, |
| "clip_ratio/low_min": 0.0035098522901535036, |
| "clip_ratio/region_mean": 0.01998605579137802, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 22.0, |
| "completions/max_terminated_length": 22.0, |
| "completions/mean_length": 11.420000076293945, |
| "completions/mean_terminated_length": 11.420000076293945, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.16217061281204223, |
| "epoch": 0.2907608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07788415998220444, |
| "learning_rate": 5e-05, |
| "loss": 0.0008381896186619997, |
| "num_tokens": 5223674.0, |
| "reward": -0.005576171912252903, |
| "reward_std": 0.002125142840668559, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.005576171912252903, |
| "rewards/length_penalty/std": 0.002125142840668559, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9772226214408875, |
| "sampling/importance_sampling_ratio/min": 0.036092452704906464, |
| "sampling/sampling_logp_difference/max": 3.321671485900879, |
| "sampling/sampling_logp_difference/mean": 0.10923026502132416, |
| "step": 107, |
| "step_time": 1.4493822269141674 |
| }, |
| { |
| "clip_ratio/high_max": 0.001587301678955555, |
| "clip_ratio/high_mean": 0.001587301678955555, |
| "clip_ratio/low_mean": 0.0030798389576375484, |
| "clip_ratio/low_min": 0.0030798389576375484, |
| "clip_ratio/region_mean": 0.004667140636593103, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 19.0, |
| "completions/max_terminated_length": 19.0, |
| "completions/mean_length": 12.460000038146973, |
| "completions/mean_terminated_length": 12.460000038146973, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.1335143819451332, |
| "epoch": 0.29347826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023423444479703903, |
| "learning_rate": 5e-05, |
| "loss": 0.00038561897235922515, |
| "num_tokens": 5227347.0, |
| "reward": 0.2339160144329071, |
| "reward_std": 0.43201112747192383, |
| "rewards/correctness/mean": 0.23999999463558197, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.0060839843936264515, |
| "rewards/length_penalty/std": 0.0015943435719236732, |
| "sampling/importance_sampling_ratio/max": 2.6867942810058594, |
| "sampling/importance_sampling_ratio/mean": 0.9941668510437012, |
| "sampling/importance_sampling_ratio/min": 0.21056081354618073, |
| "sampling/sampling_logp_difference/max": 1.5579807758331299, |
| "sampling/sampling_logp_difference/mean": 0.03585873171687126, |
| "step": 108, |
| "step_time": 1.3730515560600907 |
| }, |
| { |
| "clip_ratio/high_max": 0.0017241379246115685, |
| "clip_ratio/high_mean": 0.0017241379246115685, |
| "clip_ratio/low_mean": 0.0035725677385926246, |
| "clip_ratio/low_min": 0.0035725677385926246, |
| "clip_ratio/region_mean": 0.0052967056632041935, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 11.559999465942383, |
| "completions/mean_terminated_length": 11.559999465942383, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.1513924017548561, |
| "epoch": 0.296195652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024015560746192932, |
| "learning_rate": 5e-05, |
| "loss": 0.0002973989467136562, |
| "num_tokens": 5230265.0, |
| "reward": 0.17435546219348907, |
| "reward_std": 0.3879881203174591, |
| "rewards/correctness/mean": 0.18000000715255737, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.005644531454890966, |
| "rewards/length_penalty/std": 0.0012047950876876712, |
| "sampling/importance_sampling_ratio/max": 2.225109100341797, |
| "sampling/importance_sampling_ratio/mean": 0.9949972629547119, |
| "sampling/importance_sampling_ratio/min": 0.22619250416755676, |
| "sampling/sampling_logp_difference/max": 1.4863688945770264, |
| "sampling/sampling_logp_difference/mean": 0.036962445825338364, |
| "step": 109, |
| "step_time": 1.3599401952233166 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 25.0, |
| "completions/max_terminated_length": 25.0, |
| "completions/mean_length": 10.739999771118164, |
| "completions/mean_terminated_length": 10.739999771118164, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.11147271394729615, |
| "epoch": 0.29891304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05641806498169899, |
| "learning_rate": 5e-05, |
| "loss": 0.0008380092331208289, |
| "num_tokens": 5233382.0, |
| "reward": -0.005244140513241291, |
| "reward_std": 0.001527638640254736, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.005244140513241291, |
| "rewards/length_penalty/std": 0.0015276387566700578, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0056438446044922, |
| "sampling/importance_sampling_ratio/min": 0.5855758190155029, |
| "sampling/sampling_logp_difference/max": 1.4258408546447754, |
| "sampling/sampling_logp_difference/mean": 0.03065583109855652, |
| "step": 110, |
| "step_time": 1.417043400928378 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.006247987225651741, |
| "clip_ratio/low_min": 0.006247987225651741, |
| "clip_ratio/region_mean": 0.006247987225651741, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 9.639999389648438, |
| "completions/mean_terminated_length": 9.639999389648438, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.10073150843381881, |
| "epoch": 0.3016304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014634773135185242, |
| "learning_rate": 5e-05, |
| "loss": 0.00038955200579948723, |
| "num_tokens": 5236524.0, |
| "reward": 0.37529295682907104, |
| "reward_std": 0.48980987071990967, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.004707031417638063, |
| "rewards/length_penalty/std": 0.00078149902401492, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9968447089195251, |
| "sampling/importance_sampling_ratio/min": 0.3048332929611206, |
| "sampling/sampling_logp_difference/max": 1.475881576538086, |
| "sampling/sampling_logp_difference/mean": 0.02964521385729313, |
| "step": 111, |
| "step_time": 1.336422686232254 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016129031777381898, |
| "clip_ratio/high_mean": 0.0016129031777381898, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0016129031777381898, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 24.0, |
| "completions/max_terminated_length": 24.0, |
| "completions/mean_length": 13.59999942779541, |
| "completions/mean_terminated_length": 13.59999942779541, |
| "completions/min_length": 10.0, |
| "completions/min_terminated_length": 10.0, |
| "entropy": 0.10156970024108887, |
| "epoch": 0.30434782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012489481829106808, |
| "learning_rate": 5e-05, |
| "loss": 0.0008439509547315538, |
| "num_tokens": 5239504.0, |
| "reward": 0.193359375, |
| "reward_std": 0.40347129106521606, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.006640625186264515, |
| "rewards/length_penalty/std": 0.0014498193049803376, |
| "sampling/importance_sampling_ratio/max": 1.178148627281189, |
| "sampling/importance_sampling_ratio/mean": 0.9972023367881775, |
| "sampling/importance_sampling_ratio/min": 0.8201668858528137, |
| "sampling/sampling_logp_difference/max": 0.19824743270874023, |
| "sampling/sampling_logp_difference/mean": 0.006902283988893032, |
| "step": 112, |
| "step_time": 1.3959548557177186 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0013333333656191826, |
| "clip_ratio/low_min": 0.0013333333656191826, |
| "clip_ratio/region_mean": 0.0013333333656191826, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 26.0, |
| "completions/max_terminated_length": 26.0, |
| "completions/mean_length": 14.25999927520752, |
| "completions/mean_terminated_length": 14.25999927520752, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.16349268555641175, |
| "epoch": 0.3070652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022134723141789436, |
| "learning_rate": 5e-05, |
| "loss": 0.0006864924216642976, |
| "num_tokens": 5243527.0, |
| "reward": 0.13303710520267487, |
| "reward_std": 0.35014674067497253, |
| "rewards/correctness/mean": 0.14000000059604645, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.006962890736758709, |
| "rewards/length_penalty/std": 0.0028757285326719284, |
| "sampling/importance_sampling_ratio/max": 1.4776500463485718, |
| "sampling/importance_sampling_ratio/mean": 0.9908596277236938, |
| "sampling/importance_sampling_ratio/min": 0.016920460388064384, |
| "sampling/sampling_logp_difference/max": 4.0792317390441895, |
| "sampling/sampling_logp_difference/mean": 0.03773344308137894, |
| "step": 113, |
| "step_time": 1.4314817560371011 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 10.039999961853027, |
| "completions/mean_terminated_length": 10.039999961853027, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.061382395774126054, |
| "epoch": 0.30978260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021512916311621666, |
| "learning_rate": 5e-05, |
| "loss": 2.3743483325233683e-05, |
| "num_tokens": 5246569.0, |
| "reward": 0.17509765923023224, |
| "reward_std": 0.3876357078552246, |
| "rewards/correctness/mean": 0.18000000715255737, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.0049023437313735485, |
| "rewards/length_penalty/std": 0.0008710094843991101, |
| "sampling/importance_sampling_ratio/max": 1.4663383960723877, |
| "sampling/importance_sampling_ratio/mean": 0.9888728857040405, |
| "sampling/importance_sampling_ratio/min": 0.4582608640193939, |
| "sampling/sampling_logp_difference/max": 0.7803167104721069, |
| "sampling/sampling_logp_difference/mean": 0.020492443814873695, |
| "step": 114, |
| "step_time": 1.3372664658818394 |
| }, |
| { |
| "clip_ratio/high_max": 0.007807972840964794, |
| "clip_ratio/high_mean": 0.007807972840964794, |
| "clip_ratio/low_mean": 0.001980197988450527, |
| "clip_ratio/low_min": 0.001980197988450527, |
| "clip_ratio/region_mean": 0.009788171015679836, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14.0, |
| "completions/max_terminated_length": 14.0, |
| "completions/mean_length": 10.800000190734863, |
| "completions/mean_terminated_length": 10.800000190734863, |
| "completions/min_length": 6.0, |
| "completions/min_terminated_length": 6.0, |
| "entropy": 0.15391483157873154, |
| "epoch": 0.3125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03519968315958977, |
| "learning_rate": 5e-05, |
| "loss": 0.00037652882747352123, |
| "num_tokens": 5249339.0, |
| "reward": 0.05472655966877937, |
| "reward_std": 0.240000918507576, |
| "rewards/correctness/mean": 0.05999999865889549, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.0052734375931322575, |
| "rewards/length_penalty/std": 0.0012594496365636587, |
| "sampling/importance_sampling_ratio/max": 2.970386266708374, |
| "sampling/importance_sampling_ratio/mean": 0.9945979118347168, |
| "sampling/importance_sampling_ratio/min": 0.4335346519947052, |
| "sampling/sampling_logp_difference/max": 1.0886919498443604, |
| "sampling/sampling_logp_difference/mean": 0.05755684897303581, |
| "step": 115, |
| "step_time": 1.795232417061925 |
| }, |
| { |
| "clip_ratio/high_max": 0.002222222276031971, |
| "clip_ratio/high_mean": 0.002222222276031971, |
| "clip_ratio/low_mean": 0.002083333395421505, |
| "clip_ratio/low_min": 0.002083333395421505, |
| "clip_ratio/region_mean": 0.004305555671453476, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 8.920000076293945, |
| "completions/mean_terminated_length": 8.920000076293945, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.1007193997502327, |
| "epoch": 0.31521739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.036852993071079254, |
| "learning_rate": 5e-05, |
| "loss": 0.00039221873157657683, |
| "num_tokens": 5252675.0, |
| "reward": -0.004355468787252903, |
| "reward_std": 0.0007170813041739166, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.004355468787252903, |
| "rewards/length_penalty/std": 0.0007170813623815775, |
| "sampling/importance_sampling_ratio/max": 2.797285795211792, |
| "sampling/importance_sampling_ratio/mean": 0.9947583079338074, |
| "sampling/importance_sampling_ratio/min": 0.4579090178012848, |
| "sampling/sampling_logp_difference/max": 1.0286495685577393, |
| "sampling/sampling_logp_difference/mean": 0.04321431741118431, |
| "step": 116, |
| "step_time": 1.324604013003409 |
| }, |
| { |
| "clip_ratio/high_max": 0.009090154804289341, |
| "clip_ratio/high_mean": 0.009090154804289341, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.009090154804289341, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 31.0, |
| "completions/max_terminated_length": 31.0, |
| "completions/mean_length": 9.619999885559082, |
| "completions/mean_terminated_length": 9.619999885559082, |
| "completions/min_length": 6.0, |
| "completions/min_terminated_length": 6.0, |
| "entropy": 0.11697450280189514, |
| "epoch": 0.3179347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.035146910697221756, |
| "learning_rate": 5e-05, |
| "loss": 0.0009425780735909939, |
| "num_tokens": 5255336.0, |
| "reward": 0.03530273213982582, |
| "reward_std": 0.19812369346618652, |
| "rewards/correctness/mean": 0.03999999910593033, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.0046972655691206455, |
| "rewards/length_penalty/std": 0.002178953494876623, |
| "sampling/importance_sampling_ratio/max": 2.914954423904419, |
| "sampling/importance_sampling_ratio/mean": 1.005454421043396, |
| "sampling/importance_sampling_ratio/min": 0.19642242789268494, |
| "sampling/sampling_logp_difference/max": 1.6274876594543457, |
| "sampling/sampling_logp_difference/mean": 0.07641458511352539, |
| "step": 117, |
| "step_time": 1.4002676797099411 |
| }, |
| { |
| "clip_ratio/high_max": 0.00470588244497776, |
| "clip_ratio/high_mean": 0.00470588244497776, |
| "clip_ratio/low_mean": 0.006158198229968548, |
| "clip_ratio/low_min": 0.006158198229968548, |
| "clip_ratio/region_mean": 0.010864080674946309, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14.0, |
| "completions/max_terminated_length": 14.0, |
| "completions/mean_length": 9.239999771118164, |
| "completions/mean_terminated_length": 9.239999771118164, |
| "completions/min_length": 5.0, |
| "completions/min_terminated_length": 5.0, |
| "entropy": 0.10962557941675186, |
| "epoch": 0.32065217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.039387039840221405, |
| "learning_rate": 5e-05, |
| "loss": 0.0005441634566523135, |
| "num_tokens": 5259478.0, |
| "reward": -0.004511718638241291, |
| "reward_std": 0.0013437832240015268, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.004511718638241291, |
| "rewards/length_penalty/std": 0.001343783107586205, |
| "sampling/importance_sampling_ratio/max": 1.5118266344070435, |
| "sampling/importance_sampling_ratio/mean": 0.9934661984443665, |
| "sampling/importance_sampling_ratio/min": 0.29497018456459045, |
| "sampling/sampling_logp_difference/max": 1.2208809852600098, |
| "sampling/sampling_logp_difference/mean": 0.04646340012550354, |
| "step": 118, |
| "step_time": 1.4093847225885838 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.004662956111133099, |
| "clip_ratio/low_min": 0.004662956111133099, |
| "clip_ratio/region_mean": 0.004662956111133099, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 70.0, |
| "completions/max_terminated_length": 70.0, |
| "completions/mean_length": 17.299999237060547, |
| "completions/mean_terminated_length": 17.299999237060547, |
| "completions/min_length": 6.0, |
| "completions/min_terminated_length": 6.0, |
| "entropy": 0.20006234347820281, |
| "epoch": 0.3233695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03536223620176315, |
| "learning_rate": 5e-05, |
| "loss": 0.0027659833431243896, |
| "num_tokens": 5263853.0, |
| "reward": -0.00844726525247097, |
| "reward_std": 0.008992240764200687, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00844726525247097, |
| "rewards/length_penalty/std": 0.008992240764200687, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9917269945144653, |
| "sampling/importance_sampling_ratio/min": 0.1132407858967781, |
| "sampling/sampling_logp_difference/max": 2.178238868713379, |
| "sampling/sampling_logp_difference/mean": 0.059920959174633026, |
| "step": 119, |
| "step_time": 1.69672428118065 |
| }, |
| { |
| "clip_ratio/high_max": 0.012877547927200795, |
| "clip_ratio/high_mean": 0.012877547927200795, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.012877547927200795, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10.0, |
| "completions/max_terminated_length": 10.0, |
| "completions/mean_length": 6.659999847412109, |
| "completions/mean_terminated_length": 6.659999847412109, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.22790044248104097, |
| "epoch": 0.32608695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011693731881678104, |
| "learning_rate": 5e-05, |
| "loss": 0.00021851649216841906, |
| "num_tokens": 5267036.0, |
| "reward": -0.003251953050494194, |
| "reward_std": 0.0008347834809683263, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.003251953050494194, |
| "rewards/length_penalty/std": 0.0008347834227606654, |
| "sampling/importance_sampling_ratio/max": 1.4533648490905762, |
| "sampling/importance_sampling_ratio/mean": 0.9880709648132324, |
| "sampling/importance_sampling_ratio/min": 0.4679463803768158, |
| "sampling/sampling_logp_difference/max": 0.7594015598297119, |
| "sampling/sampling_logp_difference/mean": 0.02989094890654087, |
| "step": 120, |
| "step_time": 1.27832273975946 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6.0, |
| "completions/max_terminated_length": 6.0, |
| "completions/mean_length": 6.0, |
| "completions/mean_terminated_length": 6.0, |
| "completions/min_length": 6.0, |
| "completions/min_terminated_length": 6.0, |
| "entropy": 0.026507816463708877, |
| "epoch": 0.328804347826087, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5270676.0, |
| "reward": -0.0029296875, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0029296875, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2839123010635376, |
| "sampling/importance_sampling_ratio/mean": 0.9983105659484863, |
| "sampling/importance_sampling_ratio/min": 0.8477827310562134, |
| "sampling/sampling_logp_difference/max": 0.24991190433502197, |
| "sampling/sampling_logp_difference/mean": 0.004507859703153372, |
| "step": 121, |
| "step_time": 1.283884373260662 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8.0, |
| "completions/max_terminated_length": 8.0, |
| "completions/mean_length": 6.119999885559082, |
| "completions/mean_terminated_length": 6.119999885559082, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.08111751973628997, |
| "epoch": 0.33152173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.04897897690534592, |
| "learning_rate": 5e-05, |
| "loss": 0.00029342208290472627, |
| "num_tokens": 5274102.0, |
| "reward": -0.0029882811941206455, |
| "reward_std": 0.0005971243372187018, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0029882811941206455, |
| "rewards/length_penalty/std": 0.0005971242790110409, |
| "sampling/importance_sampling_ratio/max": 1.290226697921753, |
| "sampling/importance_sampling_ratio/mean": 0.9937352538108826, |
| "sampling/importance_sampling_ratio/min": 0.5081766247749329, |
| "sampling/sampling_logp_difference/max": 0.6769261360168457, |
| "sampling/sampling_logp_difference/mean": 0.019162509590387344, |
| "step": 122, |
| "step_time": 1.3083917561452836 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.002985074557363987, |
| "clip_ratio/low_min": 0.002985074557363987, |
| "clip_ratio/region_mean": 0.002985074557363987, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9.0, |
| "completions/max_terminated_length": 9.0, |
| "completions/mean_length": 5.779999732971191, |
| "completions/mean_terminated_length": 5.779999732971191, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.13758472353219986, |
| "epoch": 0.3342391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.041824474930763245, |
| "learning_rate": 5e-05, |
| "loss": 0.0002840502711478621, |
| "num_tokens": 5277631.0, |
| "reward": -0.0028222654946148396, |
| "reward_std": 0.000726853555534035, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.002822265727445483, |
| "rewards/length_penalty/std": 0.0007268536137416959, |
| "sampling/importance_sampling_ratio/max": 1.4336868524551392, |
| "sampling/importance_sampling_ratio/mean": 0.985608696937561, |
| "sampling/importance_sampling_ratio/min": 0.7010058164596558, |
| "sampling/sampling_logp_difference/max": 0.36024928092956543, |
| "sampling/sampling_logp_difference/mean": 0.025418559089303017, |
| "step": 123, |
| "step_time": 1.2683060599956661 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5.0, |
| "completions/max_terminated_length": 5.0, |
| "completions/mean_length": 4.199999809265137, |
| "completions/mean_terminated_length": 4.199999809265137, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.005000581219792366, |
| "epoch": 0.33695652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5280091.0, |
| "reward": -0.0020507811568677425, |
| "reward_std": 0.00019729541963897645, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0020507811568677425, |
| "rewards/length_penalty/std": 0.00019729541963897645, |
| "sampling/importance_sampling_ratio/max": 1.0457961559295654, |
| "sampling/importance_sampling_ratio/mean": 1.0017213821411133, |
| "sampling/importance_sampling_ratio/min": 0.99504554271698, |
| "sampling/sampling_logp_difference/max": 0.04477844759821892, |
| "sampling/sampling_logp_difference/mean": 0.0023150492925196886, |
| "step": 124, |
| "step_time": 1.2405255211051553 |
| }, |
| { |
| "clip_ratio/high_max": 0.0037735849618911743, |
| "clip_ratio/high_mean": 0.0037735849618911743, |
| "clip_ratio/low_mean": 0.0015625, |
| "clip_ratio/low_min": 0.0015625, |
| "clip_ratio/region_mean": 0.005336084961891174, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 26.0, |
| "completions/max_terminated_length": 26.0, |
| "completions/mean_length": 8.420000076293945, |
| "completions/mean_terminated_length": 8.420000076293945, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.07329838648438454, |
| "epoch": 0.33967391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012392704375088215, |
| "learning_rate": 5e-05, |
| "loss": 0.000670652836561203, |
| "num_tokens": 5282992.0, |
| "reward": -0.004111328162252903, |
| "reward_std": 0.0036206396762281656, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.004111328162252903, |
| "rewards/length_penalty/std": 0.0036206399090588093, |
| "sampling/importance_sampling_ratio/max": 1.699129343032837, |
| "sampling/importance_sampling_ratio/mean": 0.9851216673851013, |
| "sampling/importance_sampling_ratio/min": 0.11452800035476685, |
| "sampling/sampling_logp_difference/max": 2.166935920715332, |
| "sampling/sampling_logp_difference/mean": 0.05002863332629204, |
| "step": 125, |
| "step_time": 1.3713963988702744 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6.0, |
| "completions/max_terminated_length": 6.0, |
| "completions/mean_length": 4.799999713897705, |
| "completions/mean_terminated_length": 4.799999713897705, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.08750550970435142, |
| "epoch": 0.3423913043478261, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5288202.0, |
| "reward": -0.002343749860301614, |
| "reward_std": 0.0004832730919588357, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0023437500931322575, |
| "rewards/length_penalty/std": 0.0004832730919588357, |
| "sampling/importance_sampling_ratio/max": 1.0783474445343018, |
| "sampling/importance_sampling_ratio/mean": 0.9900667071342468, |
| "sampling/importance_sampling_ratio/min": 0.7434138059616089, |
| "sampling/sampling_logp_difference/max": 0.29650241136550903, |
| "sampling/sampling_logp_difference/mean": 0.011957721784710884, |
| "step": 126, |
| "step_time": 1.7180113978683949 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 4.0, |
| "completions/mean_terminated_length": 4.0, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.03448439426720142, |
| "epoch": 0.3451086956521739, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5291032.0, |
| "reward": -0.001953125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.001953125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3445651531219482, |
| "sampling/importance_sampling_ratio/mean": 0.989726185798645, |
| "sampling/importance_sampling_ratio/min": 0.5490431189537048, |
| "sampling/sampling_logp_difference/max": 0.5995782613754272, |
| "sampling/sampling_logp_difference/mean": 0.01975761540234089, |
| "step": 127, |
| "step_time": 1.2396131267305464 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 4.0, |
| "completions/mean_terminated_length": 4.0, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.0007371076295385137, |
| "epoch": 0.34782608695652173, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5293552.0, |
| "reward": -0.001953125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.001953125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999183416366577, |
| "sampling/importance_sampling_ratio/min": 0.99826580286026, |
| "sampling/sampling_logp_difference/max": 0.001735687255859375, |
| "sampling/sampling_logp_difference/mean": 8.172988600563258e-05, |
| "step": 128, |
| "step_time": 1.2522027201484889 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6.0, |
| "completions/max_terminated_length": 6.0, |
| "completions/mean_length": 4.400000095367432, |
| "completions/mean_terminated_length": 4.400000095367432, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.06384843587875366, |
| "epoch": 0.35054347826086957, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5296832.0, |
| "reward": -0.0021484375465661287, |
| "reward_std": 0.0003945908392779529, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0021484375465661287, |
| "rewards/length_penalty/std": 0.0003945908392779529, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0184239149093628, |
| "sampling/importance_sampling_ratio/min": 0.7039918303489685, |
| "sampling/sampling_logp_difference/max": 1.1945016384124756, |
| "sampling/sampling_logp_difference/mean": 0.02618381939828396, |
| "step": 129, |
| "step_time": 1.263973186723888 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 4.0, |
| "completions/mean_terminated_length": 4.0, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.000774494803044945, |
| "epoch": 0.3532608695652174, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5299802.0, |
| "reward": -0.001953125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.001953125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999057054519653, |
| "sampling/importance_sampling_ratio/min": 0.998631477355957, |
| "sampling/sampling_logp_difference/max": 0.001369476318359375, |
| "sampling/sampling_logp_difference/mean": 9.437560947844759e-05, |
| "step": 130, |
| "step_time": 1.2340265030506998 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 4.0, |
| "completions/mean_terminated_length": 4.0, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.03302949406206608, |
| "epoch": 0.35597826086956524, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5303382.0, |
| "reward": -0.001953125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.001953125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5545979738235474, |
| "sampling/importance_sampling_ratio/mean": 0.9879531860351562, |
| "sampling/importance_sampling_ratio/min": 0.37744536995887756, |
| "sampling/sampling_logp_difference/max": 0.9743294715881348, |
| "sampling/sampling_logp_difference/mean": 0.03508816659450531, |
| "step": 131, |
| "step_time": 1.2347275740467012 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 4.0, |
| "completions/mean_terminated_length": 4.0, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.018899894878268243, |
| "epoch": 0.358695652173913, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5305412.0, |
| "reward": -0.001953125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.001953125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3826532363891602, |
| "sampling/importance_sampling_ratio/mean": 0.9887005686759949, |
| "sampling/importance_sampling_ratio/min": 0.11252295970916748, |
| "sampling/sampling_logp_difference/max": 2.184597969055176, |
| "sampling/sampling_logp_difference/mean": 0.04885903373360634, |
| "step": 132, |
| "step_time": 1.2340079478453845 |
| }, |
| { |
| "clip_ratio/high_max": 0.02052631601691246, |
| "clip_ratio/high_mean": 0.02052631601691246, |
| "clip_ratio/low_mean": 0.005263157933950424, |
| "clip_ratio/low_min": 0.005263157933950424, |
| "clip_ratio/region_mean": 0.025789473205804825, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 3.919999837875366, |
| "completions/mean_terminated_length": 3.919999837875366, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.0779256984591484, |
| "epoch": 0.36141304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.052000973373651505, |
| "learning_rate": 5e-05, |
| "loss": -1.3637029042001814e-05, |
| "num_tokens": 5308298.0, |
| "reward": -0.0019140624208375812, |
| "reward_std": 0.00019330924260430038, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.001914062537252903, |
| "rewards/length_penalty/std": 0.00019330924260430038, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9617674350738525, |
| "sampling/importance_sampling_ratio/min": 0.01242777705192566, |
| "sampling/sampling_logp_difference/max": 4.387821197509766, |
| "sampling/sampling_logp_difference/mean": 0.1869630068540573, |
| "step": 133, |
| "step_time": 1.229072863003239 |
| }, |
| { |
| "clip_ratio/high_max": 0.007692307978868484, |
| "clip_ratio/high_mean": 0.007692307978868484, |
| "clip_ratio/low_mean": 0.049358976632356645, |
| "clip_ratio/low_min": 0.049358976632356645, |
| "clip_ratio/region_mean": 0.05705128461122513, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 2.799999952316284, |
| "completions/mean_terminated_length": 2.799999952316284, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.05317694060504437, |
| "epoch": 0.3641304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017565274611115456, |
| "learning_rate": 5e-05, |
| "loss": 0.00011389278370188549, |
| "num_tokens": 5311128.0, |
| "reward": -0.0013671874767169356, |
| "reward_std": 0.0004832731210626662, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0013671874767169356, |
| "rewards/length_penalty/std": 0.0004832730919588357, |
| "sampling/importance_sampling_ratio/max": 2.4350147247314453, |
| "sampling/importance_sampling_ratio/mean": 0.9562846422195435, |
| "sampling/importance_sampling_ratio/min": 0.030706245452165604, |
| "sampling/sampling_logp_difference/max": 3.4832892417907715, |
| "sampling/sampling_logp_difference/mean": 0.1683555394411087, |
| "step": 134, |
| "step_time": 1.2204838218167424 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 2.0, |
| "completions/mean_terminated_length": 2.0, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.009394463058561087, |
| "epoch": 0.36684782608695654, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5315408.0, |
| "reward": -0.0009765625, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0009765625, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9987919926643372, |
| "sampling/importance_sampling_ratio/min": 0.9829053282737732, |
| "sampling/sampling_logp_difference/max": 0.017242431640625, |
| "sampling/sampling_logp_difference/mean": 0.0012125587090849876, |
| "step": 135, |
| "step_time": 1.276362112024799 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 2.0, |
| "completions/mean_terminated_length": 2.0, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.17275262624025345, |
| "epoch": 0.3695652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5317588.0, |
| "reward": -0.0009765625, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0009765625, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9371755719184875, |
| "sampling/importance_sampling_ratio/min": 0.28279340267181396, |
| "sampling/sampling_logp_difference/max": 1.2630386352539062, |
| "sampling/sampling_logp_difference/mean": 0.10002973675727844, |
| "step": 136, |
| "step_time": 1.2192977210506797 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 2.0, |
| "completions/mean_terminated_length": 2.0, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.09764315634965896, |
| "epoch": 0.37228260869565216, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5320548.0, |
| "reward": -0.0009765625, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0009765625, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9831224083900452, |
| "sampling/importance_sampling_ratio/min": 0.8694930672645569, |
| "sampling/sampling_logp_difference/max": 0.1398448944091797, |
| "sampling/sampling_logp_difference/mean": 0.017522353678941727, |
| "step": 137, |
| "step_time": 1.227608114015311 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.01111111119389534, |
| "clip_ratio/low_min": 0.01111111119389534, |
| "clip_ratio/region_mean": 0.01111111119389534, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 1.8600000143051147, |
| "completions/mean_terminated_length": 1.8600000143051147, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 0.04910976588726044, |
| "epoch": 0.375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0014119825791567564, |
| "learning_rate": 5e-05, |
| "loss": -6.355220830300823e-05, |
| "num_tokens": 5323671.0, |
| "reward": -0.0009082031319849193, |
| "reward_std": 0.00017114737420342863, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0009082031319849193, |
| "rewards/length_penalty/std": 0.00017114737420342863, |
| "sampling/importance_sampling_ratio/max": 1.6528080701828003, |
| "sampling/importance_sampling_ratio/mean": 1.0031979084014893, |
| "sampling/importance_sampling_ratio/min": 0.002870837925001979, |
| "sampling/sampling_logp_difference/max": 5.853151321411133, |
| "sampling/sampling_logp_difference/mean": 0.21705520153045654, |
| "step": 138, |
| "step_time": 1.5940347050782293 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.07728058099746704, |
| "clip_ratio/low_min": 0.07728058099746704, |
| "clip_ratio/region_mean": 0.07728058099746704, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 1.5999999046325684, |
| "completions/mean_terminated_length": 1.5999999046325684, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 0.07732168436050416, |
| "epoch": 0.37771739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009142358787357807, |
| "learning_rate": 5e-05, |
| "loss": -0.00011941399134229869, |
| "num_tokens": 5326531.0, |
| "reward": -0.0007812499534338713, |
| "reward_std": 0.0002416365605313331, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0007812500116415322, |
| "rewards/length_penalty/std": 0.00024163654597941786, |
| "sampling/importance_sampling_ratio/max": 2.6340532302856445, |
| "sampling/importance_sampling_ratio/mean": 0.8506826758384705, |
| "sampling/importance_sampling_ratio/min": 0.004517478868365288, |
| "sampling/sampling_logp_difference/max": 5.399801254272461, |
| "sampling/sampling_logp_difference/mean": 0.8341909646987915, |
| "step": 139, |
| "step_time": 1.225201359251514 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 1.6200000047683716, |
| "completions/mean_terminated_length": 1.6200000047683716, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 0.04216942563652992, |
| "epoch": 0.3804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0020231842063367367, |
| "learning_rate": 5e-05, |
| "loss": -7.530484253948089e-06, |
| "num_tokens": 5330252.0, |
| "reward": -0.0007910156273283064, |
| "reward_std": 0.00023941129620652646, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0007910156273283064, |
| "rewards/length_penalty/std": 0.00023941129620652646, |
| "sampling/importance_sampling_ratio/max": 1.1531181335449219, |
| "sampling/importance_sampling_ratio/mean": 0.9962000250816345, |
| "sampling/importance_sampling_ratio/min": 0.04073739051818848, |
| "sampling/sampling_logp_difference/max": 3.200608968734741, |
| "sampling/sampling_logp_difference/mean": 0.06219974532723427, |
| "step": 140, |
| "step_time": 1.2356867531780154 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.06969697177410125, |
| "clip_ratio/low_min": 0.06969697177410125, |
| "clip_ratio/region_mean": 0.06969697177410125, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 1.1200000047683716, |
| "completions/mean_terminated_length": 1.1200000047683716, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 0.012958467192947864, |
| "epoch": 0.38315217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0012079054722562432, |
| "learning_rate": 5e-05, |
| "loss": -5.895784852327779e-05, |
| "num_tokens": 5333038.0, |
| "reward": -0.0005468750023283064, |
| "reward_std": 0.00016028355457819998, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0005468750023283064, |
| "rewards/length_penalty/std": 0.00016028355457819998, |
| "sampling/importance_sampling_ratio/max": 1.6550384759902954, |
| "sampling/importance_sampling_ratio/mean": 1.0014625787734985, |
| "sampling/importance_sampling_ratio/min": 0.008147133514285088, |
| "sampling/sampling_logp_difference/max": 4.810089111328125, |
| "sampling/sampling_logp_difference/mean": 0.42696085572242737, |
| "step": 141, |
| "step_time": 1.1936538128647953 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2.0, |
| "completions/max_terminated_length": 2.0, |
| "completions/mean_length": 1.1999999284744263, |
| "completions/mean_terminated_length": 1.1999999284744263, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 0.004924666986335069, |
| "epoch": 0.3858695652173913, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5335888.0, |
| "reward": -0.0005859374650754035, |
| "reward_std": 0.00019729541963897645, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0005859375232830644, |
| "rewards/length_penalty/std": 0.00019729541963897645, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9991536736488342, |
| "sampling/importance_sampling_ratio/min": 0.9933048486709595, |
| "sampling/sampling_logp_difference/max": 0.006717681884765625, |
| "sampling/sampling_logp_difference/mean": 0.00084857945330441, |
| "step": 142, |
| "step_time": 1.2055782845709473 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 5.702274597751966e-06, |
| "epoch": 0.38858695652173914, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5338278.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999998807907104, |
| "sampling/importance_sampling_ratio/min": 0.9999962449073792, |
| "sampling/sampling_logp_difference/max": 3.814697265625e-06, |
| "sampling/sampling_logp_difference/mean": 7.629394360719743e-08, |
| "step": 143, |
| "step_time": 1.1913842742796987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.275814904351137e-06, |
| "epoch": 0.391304347826087, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5341378.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999997019767761, |
| "sampling/importance_sampling_ratio/min": 0.9999847412109375, |
| "sampling/sampling_logp_difference/max": 1.52587890625e-05, |
| "sampling/sampling_logp_difference/mean": 3.0517577442878974e-07, |
| "step": 144, |
| "step_time": 1.2058883712161332 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.848629944513959e-06, |
| "epoch": 0.39402173913043476, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5343758.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999998807907104, |
| "sampling/importance_sampling_ratio/min": 0.9999962449073792, |
| "sampling/sampling_logp_difference/max": 3.814697265625e-06, |
| "sampling/sampling_logp_difference/mean": 7.629394360719743e-08, |
| "step": 145, |
| "step_time": 1.2081141660455614 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 6.686406450739923e-08, |
| "epoch": 0.3967391304347826, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5346748.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 146, |
| "step_time": 1.2110468870960176 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.017903907220898e-07, |
| "epoch": 0.39945652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5349098.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 147, |
| "step_time": 1.1976523408666253 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.6070625526463117e-07, |
| "epoch": 0.40217391304347827, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5352058.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 148, |
| "step_time": 1.1862720686476678 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 4.65775016778025e-08, |
| "epoch": 0.4048913043478261, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5355078.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 149, |
| "step_time": 1.2247103170957416 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.7427707632577948e-08, |
| "epoch": 0.4076086956521739, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5356888.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 150, |
| "step_time": 1.6283493610098958 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 7.290231053502793e-07, |
| "epoch": 0.41032608695652173, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5359908.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 151, |
| "step_time": 1.242248898372054 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 6.896822846158557e-08, |
| "epoch": 0.41304347826086957, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5363138.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 152, |
| "step_time": 1.2805996688548476 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.4553229138746245e-08, |
| "epoch": 0.4157608695652174, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5366708.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 153, |
| "step_time": 1.216064278036356 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.348719303630787e-07, |
| "epoch": 0.41847826086956524, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5370518.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 154, |
| "step_time": 1.2451905668713152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 9.523449122639249e-08, |
| "epoch": 0.421195652173913, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5373598.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 155, |
| "step_time": 1.2287437003105879 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.1791185389474776e-07, |
| "epoch": 0.42391304347826086, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5377208.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 156, |
| "step_time": 1.2365406069438905 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 8.699365281472637e-08, |
| "epoch": 0.4266304347826087, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5379938.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 157, |
| "step_time": 1.2250460148788989 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.6608468911272212e-07, |
| "epoch": 0.42934782608695654, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5383398.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 158, |
| "step_time": 1.2316680261865258 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.3573279318279673e-07, |
| "epoch": 0.4320652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5386718.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 159, |
| "step_time": 1.1855223746970296 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.299089321728843e-08, |
| "epoch": 0.43478260869565216, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5390318.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 160, |
| "step_time": 1.2253000542987138 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.5915767193064313e-08, |
| "epoch": 0.4375, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5393798.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 161, |
| "step_time": 1.7129009019117802 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.136989003531653e-08, |
| "epoch": 0.44021739130434784, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5396438.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 162, |
| "step_time": 1.2372172481846064 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.1367190470300556e-07, |
| "epoch": 0.4429347826086957, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5399198.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 163, |
| "step_time": 1.2175072208046913 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.571779387812057e-08, |
| "epoch": 0.44565217391304346, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5402108.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 164, |
| "step_time": 1.203096067532897 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.65617631864734e-05, |
| "epoch": 0.4483695652173913, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5405528.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.999998927116394, |
| "sampling/importance_sampling_ratio/min": 0.9999923706054688, |
| "sampling/sampling_logp_difference/max": 7.62939453125e-06, |
| "sampling/sampling_logp_difference/mean": 1.0681152389224735e-06, |
| "step": 165, |
| "step_time": 1.241865195799619 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 7.166194482266519e-09, |
| "epoch": 0.45108695652173914, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5407488.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 166, |
| "step_time": 1.1925839979667217 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 5.174719490241842e-07, |
| "epoch": 0.453804347826087, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5410958.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 167, |
| "step_time": 1.2339209029451013 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.5255944241232555e-08, |
| "epoch": 0.45652173913043476, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5413848.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 168, |
| "step_time": 1.2142795289400965 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 7.897388023536678e-08, |
| "epoch": 0.4592391304347826, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5416968.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 169, |
| "step_time": 1.218423033831641 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 6.477314897779252e-08, |
| "epoch": 0.46195652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5419268.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 170, |
| "step_time": 1.1932281029876322 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.8763578779612545e-08, |
| "epoch": 0.46467391304347827, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5422088.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 171, |
| "step_time": 1.2184540028683841 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 7.471353967503092e-07, |
| "epoch": 0.4673913043478261, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5425638.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 172, |
| "step_time": 1.2204364938661456 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.1194269777803357e-08, |
| "epoch": 0.4701086956521739, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5428978.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 173, |
| "step_time": 1.6243578898720443 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.9290705267092106e-08, |
| "epoch": 0.47282608695652173, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5431268.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 174, |
| "step_time": 1.2045086866710335 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.5477306476441298e-07, |
| "epoch": 0.47554347826086957, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5434398.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 175, |
| "step_time": 1.1951164260972291 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.0712519760147643e-08, |
| "epoch": 0.4782608695652174, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5437088.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 176, |
| "step_time": 1.206117364577949 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 4.687435009032015e-08, |
| "epoch": 0.48097826086956524, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5439368.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 177, |
| "step_time": 1.201211032923311 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.97534282825518e-08, |
| "epoch": 0.483695652173913, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5441878.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 178, |
| "step_time": 1.2034676268231124 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.0717691267814189e-08, |
| "epoch": 0.48641304347826086, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5444728.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 179, |
| "step_time": 1.201371184317395 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.2470789435024014e-08, |
| "epoch": 0.4891304347826087, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5449238.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 180, |
| "step_time": 1.2576132300309837 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.6236809763370276e-08, |
| "epoch": 0.49184782608695654, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5452448.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 181, |
| "step_time": 1.2096225249115378 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.38215231970662e-07, |
| "epoch": 0.4945652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5455338.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 182, |
| "step_time": 1.2098681135103106 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 6.716145946938923e-07, |
| "epoch": 0.49728260869565216, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5457678.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 183, |
| "step_time": 1.2125305379740894 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 9.269489282814903e-08, |
| "epoch": 0.5, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5460478.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 184, |
| "step_time": 1.2023981139063835 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 9.21182277124899e-08, |
| "epoch": 0.5027173913043478, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5464248.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 185, |
| "step_time": 1.622570862295106 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.8273758442433063e-06, |
| "epoch": 0.5054347826086957, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5468368.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 186, |
| "step_time": 1.223070868756622 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.3183340286104794e-08, |
| "epoch": 0.5081521739130435, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5471478.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 187, |
| "step_time": 1.2240185793489218 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 3.9594486445082566e-07, |
| "epoch": 0.5108695652173914, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5475448.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 188, |
| "step_time": 1.2653325237333775 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.631987078809516e-08, |
| "epoch": 0.5135869565217391, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5478598.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 189, |
| "step_time": 1.2116813512984663 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 9.46292004755378e-09, |
| "epoch": 0.5163043478260869, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5481208.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 190, |
| "step_time": 1.1821120167151093 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.385360488688093e-07, |
| "epoch": 0.5190217391304348, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5483568.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 191, |
| "step_time": 1.2023012631107122 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 4.1985008465417194e-07, |
| "epoch": 0.5217391304347826, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5488418.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 192, |
| "step_time": 1.299595553893596 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.989230587502334e-08, |
| "epoch": 0.5244565217391305, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5490598.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 193, |
| "step_time": 1.1842687451280653 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 5.263874527372536e-08, |
| "epoch": 0.5271739130434783, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5494308.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 194, |
| "step_time": 1.2483343563508242 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.1994130016290682e-06, |
| "epoch": 0.529891304347826, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5496718.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 195, |
| "step_time": 1.2025643151719123 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 6.655593201898569e-06, |
| "epoch": 0.532608695652174, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5499238.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 0.999999463558197, |
| "sampling/importance_sampling_ratio/min": 0.9999962449073792, |
| "sampling/sampling_logp_difference/max": 3.814697265625e-06, |
| "sampling/sampling_logp_difference/mean": 5.340576194612368e-07, |
| "step": 196, |
| "step_time": 1.195077903335914 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 1.6504539601669423e-08, |
| "epoch": 0.5353260869565217, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5502688.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 197, |
| "step_time": 1.5968098070006818 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.0920323784423546e-08, |
| "epoch": 0.5380434782608695, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5508648.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 198, |
| "step_time": 1.3996593470219523 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.6077204751118187e-08, |
| "epoch": 0.5407608695652174, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5511388.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 199, |
| "step_time": 1.1928423391655087 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1.0, |
| "completions/max_terminated_length": 1.0, |
| "completions/mean_length": 1.0, |
| "completions/mean_terminated_length": 1.0, |
| "completions/min_length": 1.0, |
| "completions/min_terminated_length": 1.0, |
| "entropy": 2.4040260093727285e-08, |
| "epoch": 0.5434782608695652, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5515698.0, |
| "reward": -0.00048828125, |
| "reward_std": 0.0, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.00048828125, |
| "rewards/length_penalty/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.0, |
| "sampling/importance_sampling_ratio/mean": 1.0, |
| "sampling/importance_sampling_ratio/min": 1.0, |
| "sampling/sampling_logp_difference/max": 0.0, |
| "sampling/sampling_logp_difference/mean": 0.0, |
| "step": 200, |
| "step_time": 1.2558702323585749 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 200, |
| "num_input_tokens_seen": 5515698, |
| "num_train_epochs": 1, |
| "save_steps": 50, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|