| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5434782608695652, |
| "eval_steps": 500, |
| "global_step": 200, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2017.0, |
| "completions/mean_length": 1671.2799072265625, |
| "completions/mean_terminated_length": 1398.4827880859375, |
| "completions/min_length": 992.0, |
| "completions/min_terminated_length": 992.0, |
| "entropy": 0.21248915791511536, |
| "epoch": 0.002717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016710713505744934, |
| "learning_rate": 0.0, |
| "loss": 0.05702488124370575, |
| "num_tokens": 86304.0, |
| "reward": -0.21605467796325684, |
| "reward_std": 0.656523585319519, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.8160547018051147, |
| "rewards/length_penalty/std": 0.18964654207229614, |
| "sampling/importance_sampling_ratio/max": 1.5222556591033936, |
| "sampling/importance_sampling_ratio/mean": 0.9927070140838623, |
| "sampling/importance_sampling_ratio/min": 0.6435883045196533, |
| "sampling/sampling_logp_difference/max": 0.44069600105285645, |
| "sampling/sampling_logp_difference/mean": 0.013988969847559929, |
| "step": 1, |
| "step_time": 27.466240296605974 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.6800000071525574, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1991.0, |
| "completions/mean_length": 1941.47998046875, |
| "completions/mean_terminated_length": 1715.125, |
| "completions/min_length": 1079.0, |
| "completions/min_terminated_length": 1079.0, |
| "entropy": 0.2905632793903351, |
| "epoch": 0.005434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017912156879901886, |
| "learning_rate": 5e-06, |
| "loss": 0.062223393470048904, |
| "num_tokens": 186618.0, |
| "reward": -0.7279882431030273, |
| "reward_std": 0.4917677640914917, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.9479882717132568, |
| "rewards/length_penalty/std": 0.10518474876880646, |
| "sampling/importance_sampling_ratio/max": 1.450933814048767, |
| "sampling/importance_sampling_ratio/mean": 0.9901063442230225, |
| "sampling/importance_sampling_ratio/min": 0.3688630163669586, |
| "sampling/sampling_logp_difference/max": 0.9973299503326416, |
| "sampling/sampling_logp_difference/mean": 0.01780563034117222, |
| "step": 2, |
| "step_time": 29.100001596147195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010529460967518389, |
| "clip_ratio/high_mean": 0.0010529460967518389, |
| "clip_ratio/low_mean": 0.00011735391963156872, |
| "clip_ratio/low_min": 0.00011735391963156872, |
| "clip_ratio/region_mean": 0.0011703000171110034, |
| "completions/clipped_ratio": 0.4599999785423279, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2006.0, |
| "completions/mean_length": 1694.1199951171875, |
| "completions/mean_terminated_length": 1392.6666259765625, |
| "completions/min_length": 809.0, |
| "completions/min_terminated_length": 809.0, |
| "entropy": 0.2701270878314972, |
| "epoch": 0.008152173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01870577409863472, |
| "learning_rate": 1e-05, |
| "loss": 0.10401780903339386, |
| "num_tokens": 273594.0, |
| "reward": -0.2272070199251175, |
| "reward_std": 0.6533970832824707, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.8272070288658142, |
| "rewards/length_penalty/std": 0.20242100954055786, |
| "sampling/importance_sampling_ratio/max": 1.4214774370193481, |
| "sampling/importance_sampling_ratio/mean": 0.9907577037811279, |
| "sampling/importance_sampling_ratio/min": 0.5610270500183105, |
| "sampling/sampling_logp_difference/max": 0.5779862403869629, |
| "sampling/sampling_logp_difference/mean": 0.01712799444794655, |
| "step": 3, |
| "step_time": 27.46721214381978 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007164249196648598, |
| "clip_ratio/high_mean": 0.0007164249196648598, |
| "clip_ratio/low_mean": 9.294202754972503e-05, |
| "clip_ratio/low_min": 9.294202754972503e-05, |
| "clip_ratio/region_mean": 0.0008093669428490102, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2014.0, |
| "completions/mean_length": 1539.8199462890625, |
| "completions/mean_terminated_length": 1379.3421630859375, |
| "completions/min_length": 558.0, |
| "completions/min_terminated_length": 558.0, |
| "entropy": 0.2479255050420761, |
| "epoch": 0.010869565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017958812415599823, |
| "learning_rate": 1.5e-05, |
| "loss": 0.05327572301030159, |
| "num_tokens": 353485.0, |
| "reward": 0.04813476279377937, |
| "reward_std": 0.5566579103469849, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.7518652081489563, |
| "rewards/length_penalty/std": 0.2128925323486328, |
| "sampling/importance_sampling_ratio/max": 1.487140417098999, |
| "sampling/importance_sampling_ratio/mean": 0.9915771484375, |
| "sampling/importance_sampling_ratio/min": 0.5447619557380676, |
| "sampling/sampling_logp_difference/max": 0.6074063777923584, |
| "sampling/sampling_logp_difference/mean": 0.01602964475750923, |
| "step": 4, |
| "step_time": 27.413475316017866 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003669463156256825, |
| "clip_ratio/high_mean": 0.0003669463156256825, |
| "clip_ratio/low_mean": 0.0001869216008344665, |
| "clip_ratio/low_min": 0.0001869216008344665, |
| "clip_ratio/region_mean": 0.0005538679310120642, |
| "completions/clipped_ratio": 0.7599999904632568, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1917.0, |
| "completions/mean_length": 1930.6199951171875, |
| "completions/mean_terminated_length": 1558.916748046875, |
| "completions/min_length": 1210.0, |
| "completions/min_terminated_length": 1210.0, |
| "entropy": 0.2713742196559906, |
| "epoch": 0.01358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01739981397986412, |
| "learning_rate": 2e-05, |
| "loss": 0.07548411935567856, |
| "num_tokens": 453726.0, |
| "reward": -0.702685534954071, |
| "reward_std": 0.5372691750526428, |
| "rewards/correctness/mean": 0.23999999463558197, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.9426855444908142, |
| "rewards/length_penalty/std": 0.1167898029088974, |
| "sampling/importance_sampling_ratio/max": 1.4476006031036377, |
| "sampling/importance_sampling_ratio/mean": 0.9907554388046265, |
| "sampling/importance_sampling_ratio/min": 0.4112430214881897, |
| "sampling/sampling_logp_difference/max": 0.88857102394104, |
| "sampling/sampling_logp_difference/mean": 0.01734250597655773, |
| "step": 5, |
| "step_time": 29.120347464689985 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005631349748000502, |
| "clip_ratio/high_mean": 0.0005631349748000502, |
| "clip_ratio/low_mean": 0.00011475403807708063, |
| "clip_ratio/low_min": 0.00011475403807708063, |
| "clip_ratio/region_mean": 0.0006778890150599181, |
| "completions/clipped_ratio": 0.5199999809265137, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1969.0, |
| "completions/mean_length": 1722.47998046875, |
| "completions/mean_terminated_length": 1369.8333740234375, |
| "completions/min_length": 853.0, |
| "completions/min_terminated_length": 853.0, |
| "entropy": 0.25884462594985963, |
| "epoch": 0.016304347826086956, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01741980016231537, |
| "learning_rate": 2.5e-05, |
| "loss": 0.10251747071743011, |
| "num_tokens": 542840.0, |
| "reward": -0.34105467796325684, |
| "reward_std": 0.6776674389839172, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.8410546779632568, |
| "rewards/length_penalty/std": 0.20479759573936462, |
| "sampling/importance_sampling_ratio/max": 1.600274920463562, |
| "sampling/importance_sampling_ratio/mean": 0.9908631443977356, |
| "sampling/importance_sampling_ratio/min": 0.6470417976379395, |
| "sampling/sampling_logp_difference/max": 0.4701753854751587, |
| "sampling/sampling_logp_difference/mean": 0.01671559549868107, |
| "step": 6, |
| "step_time": 27.61526938411407 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008430784218944609, |
| "clip_ratio/high_mean": 0.0008430784218944609, |
| "clip_ratio/low_mean": 0.0002401427336735651, |
| "clip_ratio/low_min": 0.0002401427336735651, |
| "clip_ratio/region_mean": 0.001083221146836877, |
| "completions/clipped_ratio": 0.25999999046325684, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1965.0, |
| "completions/mean_length": 1626.8800048828125, |
| "completions/mean_terminated_length": 1478.9189453125, |
| "completions/min_length": 904.0, |
| "completions/min_terminated_length": 904.0, |
| "entropy": 0.25048062205314636, |
| "epoch": 0.019021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017669498920440674, |
| "learning_rate": 3e-05, |
| "loss": 0.038972776383161545, |
| "num_tokens": 626414.0, |
| "reward": -0.054375000298023224, |
| "reward_std": 0.5831810832023621, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308748841285706, |
| "rewards/length_penalty/mean": -0.7943750023841858, |
| "rewards/length_penalty/std": 0.18618948757648468, |
| "sampling/importance_sampling_ratio/max": 1.396524429321289, |
| "sampling/importance_sampling_ratio/mean": 0.991274893283844, |
| "sampling/importance_sampling_ratio/min": 0.5936392545700073, |
| "sampling/sampling_logp_difference/max": 0.5214835405349731, |
| "sampling/sampling_logp_difference/mean": 0.01622067764401436, |
| "step": 7, |
| "step_time": 27.611676467116922 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006369256239850074, |
| "clip_ratio/high_mean": 0.0006369256239850074, |
| "clip_ratio/low_mean": 4.4381606858223674e-05, |
| "clip_ratio/low_min": 4.4381606858223674e-05, |
| "clip_ratio/region_mean": 0.0006813072308432311, |
| "completions/clipped_ratio": 0.5600000023841858, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1969.0, |
| "completions/mean_length": 1763.8399658203125, |
| "completions/mean_terminated_length": 1402.181884765625, |
| "completions/min_length": 783.0, |
| "completions/min_terminated_length": 783.0, |
| "entropy": 0.21012113690376283, |
| "epoch": 0.021739130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0189271941781044, |
| "learning_rate": 3.5e-05, |
| "loss": 0.08604700863361359, |
| "num_tokens": 716756.0, |
| "reward": -0.4012500047683716, |
| "reward_std": 0.6652357578277588, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8612499833106995, |
| "rewards/length_penalty/std": 0.19016051292419434, |
| "sampling/importance_sampling_ratio/max": 1.7601120471954346, |
| "sampling/importance_sampling_ratio/mean": 0.9926309585571289, |
| "sampling/importance_sampling_ratio/min": 0.447883278131485, |
| "sampling/sampling_logp_difference/max": 0.80322265625, |
| "sampling/sampling_logp_difference/mean": 0.013971890322864056, |
| "step": 8, |
| "step_time": 27.643733529606834 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007221051608212292, |
| "clip_ratio/high_mean": 0.0007221051608212292, |
| "clip_ratio/low_mean": 0.00010809456434799359, |
| "clip_ratio/low_min": 0.00010809456434799359, |
| "clip_ratio/region_mean": 0.0008301997091621161, |
| "completions/clipped_ratio": 0.25999999046325684, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2019.0, |
| "completions/mean_length": 1621.2999267578125, |
| "completions/mean_terminated_length": 1471.37841796875, |
| "completions/min_length": 905.0, |
| "completions/min_terminated_length": 905.0, |
| "entropy": 0.23276224732398987, |
| "epoch": 0.024456521739130436, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017220737412571907, |
| "learning_rate": 4e-05, |
| "loss": 0.061467211693525314, |
| "num_tokens": 800291.0, |
| "reward": -0.17165037989616394, |
| "reward_std": 0.5832893252372742, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.7916504144668579, |
| "rewards/length_penalty/std": 0.1827186644077301, |
| "sampling/importance_sampling_ratio/max": 1.450933814048767, |
| "sampling/importance_sampling_ratio/mean": 0.9919682145118713, |
| "sampling/importance_sampling_ratio/min": 0.6560437083244324, |
| "sampling/sampling_logp_difference/max": 0.4215278625488281, |
| "sampling/sampling_logp_difference/mean": 0.015065660700201988, |
| "step": 9, |
| "step_time": 26.90410251659341 |
| }, |
| { |
| "clip_ratio/high_max": 0.000775165727827698, |
| "clip_ratio/high_mean": 0.000775165727827698, |
| "clip_ratio/low_mean": 9.488375071668998e-05, |
| "clip_ratio/low_min": 9.488375071668998e-05, |
| "clip_ratio/region_mean": 0.0008700494654476643, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1962.0, |
| "completions/mean_length": 1432.7999267578125, |
| "completions/mean_terminated_length": 1279.0, |
| "completions/min_length": 609.0, |
| "completions/min_terminated_length": 609.0, |
| "entropy": 0.24516624212265015, |
| "epoch": 0.02717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02012091875076294, |
| "learning_rate": 4.5e-05, |
| "loss": 0.048032406717538834, |
| "num_tokens": 876751.0, |
| "reward": 0.08039062470197678, |
| "reward_std": 0.5867196321487427, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.6996093988418579, |
| "rewards/length_penalty/std": 0.22372539341449738, |
| "sampling/importance_sampling_ratio/max": 1.450929045677185, |
| "sampling/importance_sampling_ratio/mean": 0.9915328025817871, |
| "sampling/importance_sampling_ratio/min": 0.5520605444908142, |
| "sampling/sampling_logp_difference/max": 0.5940976142883301, |
| "sampling/sampling_logp_difference/mean": 0.015855105593800545, |
| "step": 10, |
| "step_time": 26.63589583686553 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006789915671106428, |
| "clip_ratio/high_mean": 0.0006789915671106428, |
| "clip_ratio/low_mean": 0.00010515841277083381, |
| "clip_ratio/low_min": 0.00010515841277083381, |
| "clip_ratio/region_mean": 0.000784149969695136, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2034.0, |
| "completions/mean_length": 1864.0999755859375, |
| "completions/mean_terminated_length": 1730.9310302734375, |
| "completions/min_length": 1247.0, |
| "completions/min_terminated_length": 1247.0, |
| "entropy": 0.24605216085910797, |
| "epoch": 0.029891304347826088, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01800241507589817, |
| "learning_rate": 5e-05, |
| "loss": 0.0682922974228859, |
| "num_tokens": 972376.0, |
| "reward": -0.3102050721645355, |
| "reward_std": 0.574580192565918, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.910205066204071, |
| "rewards/length_penalty/std": 0.1092815026640892, |
| "sampling/importance_sampling_ratio/max": 1.4554483890533447, |
| "sampling/importance_sampling_ratio/mean": 0.9913173317909241, |
| "sampling/importance_sampling_ratio/min": 0.6510153412818909, |
| "sampling/sampling_logp_difference/max": 0.42922210693359375, |
| "sampling/sampling_logp_difference/mean": 0.015794960781931877, |
| "step": 11, |
| "step_time": 27.352951725246385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005983146547805518, |
| "clip_ratio/high_mean": 0.0005983146547805518, |
| "clip_ratio/low_mean": 0.00020522345002973452, |
| "clip_ratio/low_min": 0.00020522345002973452, |
| "clip_ratio/region_mean": 0.0008035380975343287, |
| "completions/clipped_ratio": 0.5399999618530273, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1997.0, |
| "completions/mean_length": 1750.159912109375, |
| "completions/mean_terminated_length": 1400.521728515625, |
| "completions/min_length": 872.0, |
| "completions/min_terminated_length": 872.0, |
| "entropy": 0.27514544427394866, |
| "epoch": 0.03260869565217391, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018640518188476562, |
| "learning_rate": 5e-05, |
| "loss": 0.12445467710494995, |
| "num_tokens": 1062524.0, |
| "reward": -0.3945702910423279, |
| "reward_std": 0.6705499887466431, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8545703291893005, |
| "rewards/length_penalty/std": 0.18946890532970428, |
| "sampling/importance_sampling_ratio/max": 1.4538944959640503, |
| "sampling/importance_sampling_ratio/mean": 0.990461528301239, |
| "sampling/importance_sampling_ratio/min": 0.6125840544700623, |
| "sampling/sampling_logp_difference/max": 0.4900691509246826, |
| "sampling/sampling_logp_difference/mean": 0.017677756026387215, |
| "step": 12, |
| "step_time": 27.466188322287053 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006466234859544784, |
| "clip_ratio/high_mean": 0.0006466234859544784, |
| "clip_ratio/low_mean": 0.00019993836467619986, |
| "clip_ratio/low_min": 0.00019993836467619986, |
| "clip_ratio/region_mean": 0.0008465618651825934, |
| "completions/clipped_ratio": 0.35999998450279236, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2010.0, |
| "completions/mean_length": 1709.5, |
| "completions/mean_terminated_length": 1519.09375, |
| "completions/min_length": 789.0, |
| "completions/min_terminated_length": 789.0, |
| "entropy": 0.24947609603405, |
| "epoch": 0.035326086956521736, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019262662157416344, |
| "learning_rate": 5e-05, |
| "loss": 0.07544848322868347, |
| "num_tokens": 1151019.0, |
| "reward": -0.31471678614616394, |
| "reward_std": 0.6094580888748169, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.834716796875, |
| "rewards/length_penalty/std": 0.1765967309474945, |
| "sampling/importance_sampling_ratio/max": 1.841456651687622, |
| "sampling/importance_sampling_ratio/mean": 0.9911960363388062, |
| "sampling/importance_sampling_ratio/min": 0.5515028238296509, |
| "sampling/sampling_logp_difference/max": 0.610556960105896, |
| "sampling/sampling_logp_difference/mean": 0.016418898478150368, |
| "step": 13, |
| "step_time": 26.741958545753732 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009830699127633125, |
| "clip_ratio/high_mean": 0.0009830699127633125, |
| "clip_ratio/low_mean": 8.919467218220234e-05, |
| "clip_ratio/low_min": 8.919467218220234e-05, |
| "clip_ratio/region_mean": 0.0010722645907662808, |
| "completions/clipped_ratio": 0.5199999809265137, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1976.0, |
| "completions/mean_length": 1749.97998046875, |
| "completions/mean_terminated_length": 1427.125, |
| "completions/min_length": 986.0, |
| "completions/min_terminated_length": 986.0, |
| "entropy": 0.2528272271156311, |
| "epoch": 0.03804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018053650856018066, |
| "learning_rate": 5e-05, |
| "loss": 0.08140711486339569, |
| "num_tokens": 1241078.0, |
| "reward": -0.3744824230670929, |
| "reward_std": 0.6646848917007446, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.8544824123382568, |
| "rewards/length_penalty/std": 0.1807907521724701, |
| "sampling/importance_sampling_ratio/max": 1.467110276222229, |
| "sampling/importance_sampling_ratio/mean": 0.9912034869194031, |
| "sampling/importance_sampling_ratio/min": 0.5171128511428833, |
| "sampling/sampling_logp_difference/max": 0.6594941020011902, |
| "sampling/sampling_logp_difference/mean": 0.01632832922041416, |
| "step": 14, |
| "step_time": 27.562735668150708 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007815680641215294, |
| "clip_ratio/high_mean": 0.0007815680641215294, |
| "clip_ratio/low_mean": 5.2736993529833855e-05, |
| "clip_ratio/low_min": 5.2736993529833855e-05, |
| "clip_ratio/region_mean": 0.0008343050605617464, |
| "completions/clipped_ratio": 0.2800000011920929, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2043.0, |
| "completions/mean_length": 1523.5399169921875, |
| "completions/mean_terminated_length": 1319.5833740234375, |
| "completions/min_length": 868.0, |
| "completions/min_terminated_length": 868.0, |
| "entropy": 0.17416350841522216, |
| "epoch": 0.04076086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019059283658862114, |
| "learning_rate": 5e-05, |
| "loss": 0.09265540540218353, |
| "num_tokens": 1319835.0, |
| "reward": -0.22391600906848907, |
| "reward_std": 0.6178147196769714, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.7439160346984863, |
| "rewards/length_penalty/std": 0.205035001039505, |
| "sampling/importance_sampling_ratio/max": 1.4403636455535889, |
| "sampling/importance_sampling_ratio/mean": 0.9937911629676819, |
| "sampling/importance_sampling_ratio/min": 0.6058794260025024, |
| "sampling/sampling_logp_difference/max": 0.5010743141174316, |
| "sampling/sampling_logp_difference/mean": 0.011776668019592762, |
| "step": 15, |
| "step_time": 26.460759886307642 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011528952280059456, |
| "clip_ratio/high_mean": 0.0011528952280059456, |
| "clip_ratio/low_mean": 8.327624309458769e-05, |
| "clip_ratio/low_min": 8.327624309458769e-05, |
| "clip_ratio/region_mean": 0.0012361714616417885, |
| "completions/clipped_ratio": 0.4599999785423279, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2027.0, |
| "completions/mean_length": 1670.43994140625, |
| "completions/mean_terminated_length": 1348.8148193359375, |
| "completions/min_length": 764.0, |
| "completions/min_terminated_length": 764.0, |
| "entropy": 0.2705189764499664, |
| "epoch": 0.043478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019888967275619507, |
| "learning_rate": 5e-05, |
| "loss": 0.08121698349714279, |
| "num_tokens": 1407457.0, |
| "reward": -0.2756445109844208, |
| "reward_std": 0.6834150552749634, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.8156445026397705, |
| "rewards/length_penalty/std": 0.2012917548418045, |
| "sampling/importance_sampling_ratio/max": 1.439402461051941, |
| "sampling/importance_sampling_ratio/mean": 0.9905340075492859, |
| "sampling/importance_sampling_ratio/min": 0.5562198162078857, |
| "sampling/sampling_logp_difference/max": 0.5865917205810547, |
| "sampling/sampling_logp_difference/mean": 0.017149830237030983, |
| "step": 16, |
| "step_time": 26.47872799122706 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009020730736665428, |
| "clip_ratio/high_mean": 0.0009020730736665428, |
| "clip_ratio/low_mean": 5.4821876983623954e-05, |
| "clip_ratio/low_min": 5.4821876983623954e-05, |
| "clip_ratio/region_mean": 0.00095689493464306, |
| "completions/clipped_ratio": 0.6200000047683716, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2042.0, |
| "completions/mean_length": 1825.8199462890625, |
| "completions/mean_terminated_length": 1463.3157958984375, |
| "completions/min_length": 777.0, |
| "completions/min_terminated_length": 777.0, |
| "entropy": 0.24067306518554688, |
| "epoch": 0.04619565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018361249938607216, |
| "learning_rate": 5e-05, |
| "loss": 0.07143324613571167, |
| "num_tokens": 1501398.0, |
| "reward": -0.5115136504173279, |
| "reward_std": 0.641518771648407, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.8915136456489563, |
| "rewards/length_penalty/std": 0.18403741717338562, |
| "sampling/importance_sampling_ratio/max": 1.6153228282928467, |
| "sampling/importance_sampling_ratio/mean": 0.9917256236076355, |
| "sampling/importance_sampling_ratio/min": 0.590316116809845, |
| "sampling/sampling_logp_difference/max": 0.5270971059799194, |
| "sampling/sampling_logp_difference/mean": 0.01574120670557022, |
| "step": 17, |
| "step_time": 27.633284342940897 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008293627033708617, |
| "clip_ratio/high_mean": 0.0008293627033708617, |
| "clip_ratio/low_mean": 0.00015946615749271587, |
| "clip_ratio/low_min": 0.00015946615749271587, |
| "clip_ratio/region_mean": 0.0009888288448564708, |
| "completions/clipped_ratio": 0.4399999976158142, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1979.0, |
| "completions/mean_length": 1732.260009765625, |
| "completions/mean_terminated_length": 1484.1785888671875, |
| "completions/min_length": 1000.0, |
| "completions/min_terminated_length": 1000.0, |
| "entropy": 0.21614038348197936, |
| "epoch": 0.04891304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01731002889573574, |
| "learning_rate": 5e-05, |
| "loss": 0.05955525115132332, |
| "num_tokens": 1590511.0, |
| "reward": -0.30583006143569946, |
| "reward_std": 0.6507822275161743, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.8458300828933716, |
| "rewards/length_penalty/std": 0.17048171162605286, |
| "sampling/importance_sampling_ratio/max": 1.4577268362045288, |
| "sampling/importance_sampling_ratio/mean": 0.992477297782898, |
| "sampling/importance_sampling_ratio/min": 0.6387871503829956, |
| "sampling/sampling_logp_difference/max": 0.4481840133666992, |
| "sampling/sampling_logp_difference/mean": 0.014222715049982071, |
| "step": 18, |
| "step_time": 26.75244860793464 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013676838483661414, |
| "clip_ratio/high_mean": 0.0013676838483661414, |
| "clip_ratio/low_mean": 0.00011510243712109514, |
| "clip_ratio/low_min": 0.00011510243712109514, |
| "clip_ratio/region_mean": 0.0014827862847596406, |
| "completions/clipped_ratio": 0.6399999856948853, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1964.0, |
| "completions/mean_length": 1901.0, |
| "completions/mean_terminated_length": 1639.6666259765625, |
| "completions/min_length": 1197.0, |
| "completions/min_terminated_length": 1197.0, |
| "entropy": 0.29268686175346376, |
| "epoch": 0.051630434782608696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017992090433835983, |
| "learning_rate": 5e-05, |
| "loss": 0.08809743076562881, |
| "num_tokens": 1688991.0, |
| "reward": -0.5682226419448853, |
| "reward_std": 0.5851815938949585, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.92822265625, |
| "rewards/length_penalty/std": 0.11655672639608383, |
| "sampling/importance_sampling_ratio/max": 1.7099823951721191, |
| "sampling/importance_sampling_ratio/mean": 0.9895451068878174, |
| "sampling/importance_sampling_ratio/min": 0.19308936595916748, |
| "sampling/sampling_logp_difference/max": 1.6446021795272827, |
| "sampling/sampling_logp_difference/mean": 0.0185383427888155, |
| "step": 19, |
| "step_time": 28.175968994153664 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006889381911605597, |
| "clip_ratio/high_mean": 0.0006889381911605597, |
| "clip_ratio/low_mean": 0.00013566470588557422, |
| "clip_ratio/low_min": 0.00013566470588557422, |
| "clip_ratio/region_mean": 0.0008246029028669, |
| "completions/clipped_ratio": 0.5799999833106995, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1994.0, |
| "completions/mean_length": 1846.659912109375, |
| "completions/mean_terminated_length": 1568.6190185546875, |
| "completions/min_length": 1186.0, |
| "completions/min_terminated_length": 1186.0, |
| "entropy": 0.2582373857498169, |
| "epoch": 0.05434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017690176144242287, |
| "learning_rate": 5e-05, |
| "loss": 0.08645781874656677, |
| "num_tokens": 1783914.0, |
| "reward": -0.4216894507408142, |
| "reward_std": 0.6063060164451599, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.9016894698143005, |
| "rewards/length_penalty/std": 0.1410803198814392, |
| "sampling/importance_sampling_ratio/max": 1.6288467645645142, |
| "sampling/importance_sampling_ratio/mean": 0.991369903087616, |
| "sampling/importance_sampling_ratio/min": 0.5640106797218323, |
| "sampling/sampling_logp_difference/max": 0.5726821422576904, |
| "sampling/sampling_logp_difference/mean": 0.016520945355296135, |
| "step": 20, |
| "step_time": 27.12975821667351 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008033646503463387, |
| "clip_ratio/high_mean": 0.0008033646503463387, |
| "clip_ratio/low_mean": 0.00011942786586587318, |
| "clip_ratio/low_min": 0.00011942786586587318, |
| "clip_ratio/region_mean": 0.0009227925329469144, |
| "completions/clipped_ratio": 0.3799999952316284, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1991.0, |
| "completions/mean_length": 1694.0, |
| "completions/mean_terminated_length": 1477.0322265625, |
| "completions/min_length": 1037.0, |
| "completions/min_terminated_length": 1037.0, |
| "entropy": 0.23625740706920623, |
| "epoch": 0.057065217391304345, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019587965682148933, |
| "learning_rate": 5e-05, |
| "loss": 0.10707657784223557, |
| "num_tokens": 1872024.0, |
| "reward": -0.20714843273162842, |
| "reward_std": 0.6329521536827087, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.8271484375, |
| "rewards/length_penalty/std": 0.161778062582016, |
| "sampling/importance_sampling_ratio/max": 1.476094365119934, |
| "sampling/importance_sampling_ratio/mean": 0.9916335344314575, |
| "sampling/importance_sampling_ratio/min": 0.4443323314189911, |
| "sampling/sampling_logp_difference/max": 0.8111824989318848, |
| "sampling/sampling_logp_difference/mean": 0.015547686256468296, |
| "step": 21, |
| "step_time": 27.171376964775845 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006942571664694697, |
| "clip_ratio/high_mean": 0.0006942571664694697, |
| "clip_ratio/low_mean": 6.785718142054975e-05, |
| "clip_ratio/low_min": 6.785718142054975e-05, |
| "clip_ratio/region_mean": 0.0007621143420692533, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2044.0, |
| "completions/mean_length": 1486.93994140625, |
| "completions/mean_terminated_length": 1309.76318359375, |
| "completions/min_length": 693.0, |
| "completions/min_terminated_length": 693.0, |
| "entropy": 0.255102077126503, |
| "epoch": 0.059782608695652176, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019045934081077576, |
| "learning_rate": 5e-05, |
| "loss": 0.06557165086269379, |
| "num_tokens": 1951441.0, |
| "reward": 0.07395507395267487, |
| "reward_std": 0.5648945569992065, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.7260448932647705, |
| "rewards/length_penalty/std": 0.20981471240520477, |
| "sampling/importance_sampling_ratio/max": 1.579704761505127, |
| "sampling/importance_sampling_ratio/mean": 0.9910966157913208, |
| "sampling/importance_sampling_ratio/min": 0.6202152371406555, |
| "sampling/sampling_logp_difference/max": 0.47768867015838623, |
| "sampling/sampling_logp_difference/mean": 0.016288571059703827, |
| "step": 22, |
| "step_time": 26.544327920069918 |
| }, |
| { |
| "clip_ratio/high_max": 0.000928049284266308, |
| "clip_ratio/high_mean": 0.000928049284266308, |
| "clip_ratio/low_mean": 0.00011483486159704625, |
| "clip_ratio/low_min": 0.00011483486159704625, |
| "clip_ratio/region_mean": 0.0010428841284010558, |
| "completions/clipped_ratio": 0.4399999976158142, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1978.0, |
| "completions/mean_length": 1708.219970703125, |
| "completions/mean_terminated_length": 1441.2501220703125, |
| "completions/min_length": 897.0, |
| "completions/min_terminated_length": 897.0, |
| "entropy": 0.28572131395339967, |
| "epoch": 0.0625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01984834298491478, |
| "learning_rate": 5e-05, |
| "loss": 0.0938582494854927, |
| "num_tokens": 2042972.0, |
| "reward": -0.374091774225235, |
| "reward_std": 0.6551506519317627, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8340917825698853, |
| "rewards/length_penalty/std": 0.19219890236854553, |
| "sampling/importance_sampling_ratio/max": 1.5033539533615112, |
| "sampling/importance_sampling_ratio/mean": 0.9900131821632385, |
| "sampling/importance_sampling_ratio/min": 0.6327011585235596, |
| "sampling/sampling_logp_difference/max": 0.4577571153640747, |
| "sampling/sampling_logp_difference/mean": 0.018048716709017754, |
| "step": 23, |
| "step_time": 28.025786810554564 |
| }, |
| { |
| "clip_ratio/high_max": 0.00044709993817377837, |
| "clip_ratio/high_mean": 0.00044709993817377837, |
| "clip_ratio/low_mean": 0.00014162121369736268, |
| "clip_ratio/low_min": 0.00014162121369736268, |
| "clip_ratio/region_mean": 0.0005887211445951834, |
| "completions/clipped_ratio": 0.5399999618530273, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2025.0, |
| "completions/mean_length": 1685.6400146484375, |
| "completions/mean_terminated_length": 1260.2608642578125, |
| "completions/min_length": 743.0, |
| "completions/min_terminated_length": 743.0, |
| "entropy": 0.2520651489496231, |
| "epoch": 0.06521739130434782, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01798221655189991, |
| "learning_rate": 5e-05, |
| "loss": 0.09236455708742142, |
| "num_tokens": 2129544.0, |
| "reward": -0.3630664050579071, |
| "reward_std": 0.7035251259803772, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8230664134025574, |
| "rewards/length_penalty/std": 0.23066547513008118, |
| "sampling/importance_sampling_ratio/max": 1.4225741624832153, |
| "sampling/importance_sampling_ratio/mean": 0.9914146065711975, |
| "sampling/importance_sampling_ratio/min": 0.6460815668106079, |
| "sampling/sampling_logp_difference/max": 0.4368295669555664, |
| "sampling/sampling_logp_difference/mean": 0.015968643128871918, |
| "step": 24, |
| "step_time": 27.186313528334722 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007996991393156349, |
| "clip_ratio/high_mean": 0.0007996991393156349, |
| "clip_ratio/low_mean": 0.00016563651151955128, |
| "clip_ratio/low_min": 0.00016563651151955128, |
| "clip_ratio/region_mean": 0.0009653356508351862, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1904.0, |
| "completions/mean_length": 1556.5599365234375, |
| "completions/mean_terminated_length": 1303.3939208984375, |
| "completions/min_length": 658.0, |
| "completions/min_terminated_length": 658.0, |
| "entropy": 0.28897868990898135, |
| "epoch": 0.06793478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02024088054895401, |
| "learning_rate": 5e-05, |
| "loss": 0.11006991565227509, |
| "num_tokens": 2209772.0, |
| "reward": -0.1000390574336052, |
| "reward_std": 0.6651003360748291, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.7600390911102295, |
| "rewards/length_penalty/std": 0.21650709211826324, |
| "sampling/importance_sampling_ratio/max": 1.6264971494674683, |
| "sampling/importance_sampling_ratio/mean": 0.9900205731391907, |
| "sampling/importance_sampling_ratio/min": 0.6259128451347351, |
| "sampling/sampling_logp_difference/max": 0.48642873764038086, |
| "sampling/sampling_logp_difference/mean": 0.01854841411113739, |
| "step": 25, |
| "step_time": 26.23496240307577 |
| }, |
| { |
| "clip_ratio/high_max": 0.000882484030444175, |
| "clip_ratio/high_mean": 0.000882484030444175, |
| "clip_ratio/low_mean": 0.00010091252624988556, |
| "clip_ratio/low_min": 0.00010091252624988556, |
| "clip_ratio/region_mean": 0.0009833965450525284, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1743.0, |
| "completions/mean_length": 1347.239990234375, |
| "completions/mean_terminated_length": 1251.681884765625, |
| "completions/min_length": 674.0, |
| "completions/min_terminated_length": 674.0, |
| "entropy": 0.24427374601364135, |
| "epoch": 0.07065217391304347, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018950114026665688, |
| "learning_rate": 5e-05, |
| "loss": 0.07834034413099289, |
| "num_tokens": 2279604.0, |
| "reward": -0.17783202230930328, |
| "reward_std": 0.5142880082130432, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.6578320264816284, |
| "rewards/length_penalty/std": 0.18199172616004944, |
| "sampling/importance_sampling_ratio/max": 1.4621543884277344, |
| "sampling/importance_sampling_ratio/mean": 0.9915827512741089, |
| "sampling/importance_sampling_ratio/min": 0.5687382817268372, |
| "sampling/sampling_logp_difference/max": 0.5643348693847656, |
| "sampling/sampling_logp_difference/mean": 0.016337303444743156, |
| "step": 26, |
| "step_time": 25.925254836911336 |
| }, |
| { |
| "clip_ratio/high_max": 0.00110022509470582, |
| "clip_ratio/high_mean": 0.00110022509470582, |
| "clip_ratio/low_mean": 6.466519262176007e-05, |
| "clip_ratio/low_min": 6.466519262176007e-05, |
| "clip_ratio/region_mean": 0.0011648902669548987, |
| "completions/clipped_ratio": 0.4399999976158142, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1750.0, |
| "completions/mean_length": 1549.8800048828125, |
| "completions/mean_terminated_length": 1158.5, |
| "completions/min_length": 231.0, |
| "completions/min_terminated_length": 231.0, |
| "entropy": 0.21254440844058992, |
| "epoch": 0.07336956521739131, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018655436113476753, |
| "learning_rate": 5e-05, |
| "loss": 0.05900321900844574, |
| "num_tokens": 2359348.0, |
| "reward": -0.19677734375, |
| "reward_std": 0.7311728596687317, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.7567773461341858, |
| "rewards/length_penalty/std": 0.25452134013175964, |
| "sampling/importance_sampling_ratio/max": 1.6191316843032837, |
| "sampling/importance_sampling_ratio/mean": 0.9927679896354675, |
| "sampling/importance_sampling_ratio/min": 0.5498383045196533, |
| "sampling/sampling_logp_difference/max": 0.5981309413909912, |
| "sampling/sampling_logp_difference/mean": 0.01420011930167675, |
| "step": 27, |
| "step_time": 25.982643257593736 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008028354088310153, |
| "clip_ratio/high_mean": 0.0008028354088310153, |
| "clip_ratio/low_mean": 7.703569281147793e-05, |
| "clip_ratio/low_min": 7.703569281147793e-05, |
| "clip_ratio/region_mean": 0.0008798711118288338, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1910.0, |
| "completions/mean_length": 1287.239990234375, |
| "completions/mean_terminated_length": 1221.0870361328125, |
| "completions/min_length": 794.0, |
| "completions/min_terminated_length": 794.0, |
| "entropy": 0.18662778437137603, |
| "epoch": 0.07608695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01875571720302105, |
| "learning_rate": 5e-05, |
| "loss": 0.08134754747152328, |
| "num_tokens": 2427070.0, |
| "reward": 0.11146484315395355, |
| "reward_std": 0.519826352596283, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308748841285706, |
| "rewards/length_penalty/mean": -0.6285351514816284, |
| "rewards/length_penalty/std": 0.18626148998737335, |
| "sampling/importance_sampling_ratio/max": 1.6693933010101318, |
| "sampling/importance_sampling_ratio/mean": 0.9932618737220764, |
| "sampling/importance_sampling_ratio/min": 0.46988239884376526, |
| "sampling/sampling_logp_difference/max": 0.7552728652954102, |
| "sampling/sampling_logp_difference/mean": 0.013092606328427792, |
| "step": 28, |
| "step_time": 25.43532714387402 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005481465603224933, |
| "clip_ratio/high_mean": 0.0005481465603224933, |
| "clip_ratio/low_mean": 0.00013186443102313205, |
| "clip_ratio/low_min": 0.00013186443102313205, |
| "clip_ratio/region_mean": 0.0006800110160838813, |
| "completions/clipped_ratio": 0.47999998927116394, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 1640.39990234375, |
| "completions/mean_terminated_length": 1264.1539306640625, |
| "completions/min_length": 622.0, |
| "completions/min_terminated_length": 622.0, |
| "entropy": 0.26429899632930753, |
| "epoch": 0.07880434782608696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020799146965146065, |
| "learning_rate": 5e-05, |
| "loss": 0.049529045820236206, |
| "num_tokens": 2512530.0, |
| "reward": -0.26097655296325684, |
| "reward_std": 0.7104216814041138, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.800976574420929, |
| "rewards/length_penalty/std": 0.2537543773651123, |
| "sampling/importance_sampling_ratio/max": 1.9006296396255493, |
| "sampling/importance_sampling_ratio/mean": 0.9904490113258362, |
| "sampling/importance_sampling_ratio/min": 0.5602869987487793, |
| "sampling/sampling_logp_difference/max": 0.6421852111816406, |
| "sampling/sampling_logp_difference/mean": 0.017496583983302116, |
| "step": 29, |
| "step_time": 26.86644083727151 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009444889845326542, |
| "clip_ratio/high_mean": 0.0009444889845326542, |
| "clip_ratio/low_mean": 7.327521889237687e-05, |
| "clip_ratio/low_min": 7.327521889237687e-05, |
| "clip_ratio/region_mean": 0.0010177641990594566, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1965.0, |
| "completions/mean_length": 1351.919921875, |
| "completions/mean_terminated_length": 1238.6046142578125, |
| "completions/min_length": 762.0, |
| "completions/min_terminated_length": 762.0, |
| "entropy": 0.17939060628414155, |
| "epoch": 0.08152173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018544282764196396, |
| "learning_rate": 5e-05, |
| "loss": 0.09268638491630554, |
| "num_tokens": 2581916.0, |
| "reward": 0.2198828011751175, |
| "reward_std": 0.47646957635879517, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826071977615356, |
| "rewards/length_penalty/mean": -0.6601172089576721, |
| "rewards/length_penalty/std": 0.18981978297233582, |
| "sampling/importance_sampling_ratio/max": 2.3046295642852783, |
| "sampling/importance_sampling_ratio/mean": 0.9933307766914368, |
| "sampling/importance_sampling_ratio/min": 0.34033486247062683, |
| "sampling/sampling_logp_difference/max": 1.0778253078460693, |
| "sampling/sampling_logp_difference/mean": 0.01283288560807705, |
| "step": 30, |
| "step_time": 25.746895055752248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006859918124973774, |
| "clip_ratio/high_mean": 0.0006859918124973774, |
| "clip_ratio/low_mean": 0.00010212801716988907, |
| "clip_ratio/low_min": 0.00010212801716988907, |
| "clip_ratio/region_mean": 0.000788119831122458, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2034.0, |
| "completions/mean_length": 1607.679931640625, |
| "completions/mean_terminated_length": 1418.971435546875, |
| "completions/min_length": 542.0, |
| "completions/min_terminated_length": 542.0, |
| "entropy": 0.1864992558956146, |
| "epoch": 0.08423913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017994169145822525, |
| "learning_rate": 5e-05, |
| "loss": 0.06737810373306274, |
| "num_tokens": 2665430.0, |
| "reward": -0.4650000035762787, |
| "reward_std": 0.48963093757629395, |
| "rewards/correctness/mean": 0.3199999928474426, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.7850000262260437, |
| "rewards/length_penalty/std": 0.22790244221687317, |
| "sampling/importance_sampling_ratio/max": 2.577646017074585, |
| "sampling/importance_sampling_ratio/mean": 0.993366539478302, |
| "sampling/importance_sampling_ratio/min": 0.4946930408477783, |
| "sampling/sampling_logp_difference/max": 0.9468765258789062, |
| "sampling/sampling_logp_difference/mean": 0.012929298914968967, |
| "step": 31, |
| "step_time": 26.68254506518133 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005745734175434336, |
| "clip_ratio/high_mean": 0.0005745734175434336, |
| "clip_ratio/low_mean": 0.00013555512414313852, |
| "clip_ratio/low_min": 0.00013555512414313852, |
| "clip_ratio/region_mean": 0.0007101285416865721, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1663.0, |
| "completions/mean_length": 1353.93994140625, |
| "completions/mean_terminated_length": 1056.4857177734375, |
| "completions/min_length": 629.0, |
| "completions/min_terminated_length": 629.0, |
| "entropy": 0.21954194605350494, |
| "epoch": 0.08695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017363108694553375, |
| "learning_rate": 5e-05, |
| "loss": 0.08478433638811111, |
| "num_tokens": 2735057.0, |
| "reward": 0.01889648474752903, |
| "reward_std": 0.7035274505615234, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.47121208906173706, |
| "rewards/length_penalty/mean": -0.6611034870147705, |
| "rewards/length_penalty/std": 0.2549952566623688, |
| "sampling/importance_sampling_ratio/max": 2.190631151199341, |
| "sampling/importance_sampling_ratio/mean": 0.9921717047691345, |
| "sampling/importance_sampling_ratio/min": 0.30656898021698, |
| "sampling/sampling_logp_difference/max": 1.1823124885559082, |
| "sampling/sampling_logp_difference/mean": 0.015188697725534439, |
| "step": 32, |
| "step_time": 25.268918795045465 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011238815961405635, |
| "clip_ratio/high_mean": 0.0011238815961405635, |
| "clip_ratio/low_mean": 4.1938989306800066e-05, |
| "clip_ratio/low_min": 4.1938989306800066e-05, |
| "clip_ratio/region_mean": 0.0011658205767162144, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2017.0, |
| "completions/mean_length": 1405.8199462890625, |
| "completions/mean_terminated_length": 1203.0263671875, |
| "completions/min_length": 709.0, |
| "completions/min_terminated_length": 709.0, |
| "entropy": 0.20142727792263032, |
| "epoch": 0.08967391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016497349366545677, |
| "learning_rate": 5e-05, |
| "loss": 0.02186710759997368, |
| "num_tokens": 2808988.0, |
| "reward": -0.08643554151058197, |
| "reward_std": 0.6195644736289978, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716163635254, |
| "rewards/length_penalty/mean": -0.6864355206489563, |
| "rewards/length_penalty/std": 0.24275879561901093, |
| "sampling/importance_sampling_ratio/max": 2.1141459941864014, |
| "sampling/importance_sampling_ratio/mean": 0.9928951859474182, |
| "sampling/importance_sampling_ratio/min": 0.26172491908073425, |
| "sampling/sampling_logp_difference/max": 1.340461254119873, |
| "sampling/sampling_logp_difference/mean": 0.014272425323724747, |
| "step": 33, |
| "step_time": 26.138823551824316 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008198763127438724, |
| "clip_ratio/high_mean": 0.0008198763127438724, |
| "clip_ratio/low_mean": 5.119576017023064e-05, |
| "clip_ratio/low_min": 5.119576017023064e-05, |
| "clip_ratio/region_mean": 0.0008710720809176564, |
| "completions/clipped_ratio": 0.3199999928474426, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 1607.699951171875, |
| "completions/mean_terminated_length": 1400.5, |
| "completions/min_length": 871.0, |
| "completions/min_terminated_length": 871.0, |
| "entropy": 0.23463621437549592, |
| "epoch": 0.09239130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021644340828061104, |
| "learning_rate": 5e-05, |
| "loss": 0.08323785662651062, |
| "num_tokens": 2893263.0, |
| "reward": -0.1050097644329071, |
| "reward_std": 0.6311855912208557, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.7850097417831421, |
| "rewards/length_penalty/std": 0.18961191177368164, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9918093681335449, |
| "sampling/importance_sampling_ratio/min": 0.2718399167060852, |
| "sampling/sampling_logp_difference/max": 1.302541971206665, |
| "sampling/sampling_logp_difference/mean": 0.016123967245221138, |
| "step": 34, |
| "step_time": 26.62148871202953 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007957093475852162, |
| "clip_ratio/high_mean": 0.0007957093475852162, |
| "clip_ratio/low_mean": 0.00017341390484943987, |
| "clip_ratio/low_min": 0.00017341390484943987, |
| "clip_ratio/region_mean": 0.0009691232698969543, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1979.0, |
| "completions/mean_length": 1479.4000244140625, |
| "completions/mean_terminated_length": 1235.7142333984375, |
| "completions/min_length": 801.0, |
| "completions/min_terminated_length": 801.0, |
| "entropy": 0.2745778501033783, |
| "epoch": 0.09510869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02208096906542778, |
| "learning_rate": 5e-05, |
| "loss": 0.11964486539363861, |
| "num_tokens": 2969963.0, |
| "reward": -0.02236328087747097, |
| "reward_std": 0.6595257520675659, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.722363293170929, |
| "rewards/length_penalty/std": 0.22518181800842285, |
| "sampling/importance_sampling_ratio/max": 2.227848529815674, |
| "sampling/importance_sampling_ratio/mean": 0.9901186227798462, |
| "sampling/importance_sampling_ratio/min": 0.2384735494852066, |
| "sampling/sampling_logp_difference/max": 1.4334969520568848, |
| "sampling/sampling_logp_difference/mean": 0.019078632816672325, |
| "step": 35, |
| "step_time": 26.275031560100615 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003629680839367211, |
| "clip_ratio/high_mean": 0.0003629680839367211, |
| "clip_ratio/low_mean": 0.00012276005145395175, |
| "clip_ratio/low_min": 0.00012276005145395175, |
| "clip_ratio/region_mean": 0.00048572812811471523, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1757.0, |
| "completions/mean_length": 1303.5999755859375, |
| "completions/mean_terminated_length": 1117.5, |
| "completions/min_length": 669.0, |
| "completions/min_terminated_length": 669.0, |
| "entropy": 0.2164871245622635, |
| "epoch": 0.09782608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01719369739294052, |
| "learning_rate": 5e-05, |
| "loss": 0.1098581850528717, |
| "num_tokens": 3037613.0, |
| "reward": 0.16347655653953552, |
| "reward_std": 0.6018728017807007, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.636523425579071, |
| "rewards/length_penalty/std": 0.22500616312026978, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9922620058059692, |
| "sampling/importance_sampling_ratio/min": 0.2438986599445343, |
| "sampling/sampling_logp_difference/max": 1.411002516746521, |
| "sampling/sampling_logp_difference/mean": 0.01620476506650448, |
| "step": 36, |
| "step_time": 25.09496877877973 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006996341748163104, |
| "clip_ratio/high_mean": 0.0006996341748163104, |
| "clip_ratio/low_mean": 0.00016569619765505195, |
| "clip_ratio/low_min": 0.00016569619765505195, |
| "clip_ratio/region_mean": 0.0008653303724713623, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2015.0, |
| "completions/mean_length": 1335.3599853515625, |
| "completions/mean_terminated_length": 1110.3157958984375, |
| "completions/min_length": 677.0, |
| "completions/min_terminated_length": 677.0, |
| "entropy": 0.2375789314508438, |
| "epoch": 0.10054347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019913287833333015, |
| "learning_rate": 5e-05, |
| "loss": 0.07340657711029053, |
| "num_tokens": 3107901.0, |
| "reward": -0.09203124791383743, |
| "reward_std": 0.6339429616928101, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.6520312428474426, |
| "rewards/length_penalty/std": 0.23987603187561035, |
| "sampling/importance_sampling_ratio/max": 2.623735189437866, |
| "sampling/importance_sampling_ratio/mean": 0.9915706515312195, |
| "sampling/importance_sampling_ratio/min": 0.3163408637046814, |
| "sampling/sampling_logp_difference/max": 1.1509349346160889, |
| "sampling/sampling_logp_difference/mean": 0.016999898478388786, |
| "step": 37, |
| "step_time": 26.306446861475706 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006796631903853267, |
| "clip_ratio/high_mean": 0.0006796631903853267, |
| "clip_ratio/low_mean": 8.488849125569687e-05, |
| "clip_ratio/low_min": 8.488849125569687e-05, |
| "clip_ratio/region_mean": 0.0007645516889169812, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2008.0, |
| "completions/mean_length": 1376.9599609375, |
| "completions/mean_terminated_length": 1285.45458984375, |
| "completions/min_length": 749.0, |
| "completions/min_terminated_length": 749.0, |
| "entropy": 0.174486380815506, |
| "epoch": 0.10326086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018445057794451714, |
| "learning_rate": 5e-05, |
| "loss": 0.03477037698030472, |
| "num_tokens": 3180139.0, |
| "reward": 0.20765624940395355, |
| "reward_std": 0.48721736669540405, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826071977615356, |
| "rewards/length_penalty/mean": -0.6723437309265137, |
| "rewards/length_penalty/std": 0.22222350537776947, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9940947890281677, |
| "sampling/importance_sampling_ratio/min": 0.2548873722553253, |
| "sampling/sampling_logp_difference/max": 1.3669335842132568, |
| "sampling/sampling_logp_difference/mean": 0.01304861530661583, |
| "step": 38, |
| "step_time": 25.515329421730712 |
| }, |
| { |
| "clip_ratio/high_max": 0.000662170146824792, |
| "clip_ratio/high_mean": 0.000662170146824792, |
| "clip_ratio/low_mean": 0.000180268642725423, |
| "clip_ratio/low_min": 0.000180268642725423, |
| "clip_ratio/region_mean": 0.0008424387604463845, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2039.0, |
| "completions/mean_length": 1196.7999267578125, |
| "completions/mean_terminated_length": 984.0, |
| "completions/min_length": 605.0, |
| "completions/min_terminated_length": 605.0, |
| "entropy": 0.193106546998024, |
| "epoch": 0.10597826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016681300476193428, |
| "learning_rate": 5e-05, |
| "loss": 0.08455976843833923, |
| "num_tokens": 3242469.0, |
| "reward": 0.015625, |
| "reward_std": 0.7029106020927429, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.5843750238418579, |
| "rewards/length_penalty/std": 0.2486061006784439, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9931885004043579, |
| "sampling/importance_sampling_ratio/min": 0.25109171867370605, |
| "sampling/sampling_logp_difference/max": 1.381937026977539, |
| "sampling/sampling_logp_difference/mean": 0.01450310368090868, |
| "step": 39, |
| "step_time": 24.526397271314636 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005391269922256469, |
| "clip_ratio/high_mean": 0.0005391269922256469, |
| "clip_ratio/low_mean": 0.00010759711731225252, |
| "clip_ratio/low_min": 0.00010759711731225252, |
| "clip_ratio/region_mean": 0.0006467241037171334, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1342.0, |
| "completions/max_terminated_length": 1342.0, |
| "completions/mean_length": 919.5199584960938, |
| "completions/mean_terminated_length": 919.5199584960938, |
| "completions/min_length": 667.0, |
| "completions/min_terminated_length": 667.0, |
| "entropy": 0.1823535144329071, |
| "epoch": 0.10869565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01441043708473444, |
| "learning_rate": 5e-05, |
| "loss": 0.05531834438443184, |
| "num_tokens": 3291165.0, |
| "reward": 0.5510156154632568, |
| "reward_std": 0.0687033161520958, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.44898438453674316, |
| "rewards/length_penalty/std": 0.0687033161520958, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9935107231140137, |
| "sampling/importance_sampling_ratio/min": 0.2243487387895584, |
| "sampling/sampling_logp_difference/max": 1.494553565979004, |
| "sampling/sampling_logp_difference/mean": 0.015218878164887428, |
| "step": 40, |
| "step_time": 16.793462296016514 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007236653997097165, |
| "clip_ratio/high_mean": 0.0007236653997097165, |
| "clip_ratio/low_mean": 0.00017764531075954438, |
| "clip_ratio/low_min": 0.00017764531075954438, |
| "clip_ratio/region_mean": 0.0009013107162900269, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1806.0, |
| "completions/mean_length": 1353.739990234375, |
| "completions/mean_terminated_length": 1180.175048828125, |
| "completions/min_length": 660.0, |
| "completions/min_terminated_length": 660.0, |
| "entropy": 0.2105077862739563, |
| "epoch": 0.11141304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01744753308594227, |
| "learning_rate": 5e-05, |
| "loss": 0.033934399485588074, |
| "num_tokens": 3362592.0, |
| "reward": -0.20100586116313934, |
| "reward_std": 0.6444067358970642, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.6610058546066284, |
| "rewards/length_penalty/std": 0.2214658409357071, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9924235343933105, |
| "sampling/importance_sampling_ratio/min": 0.12174760550260544, |
| "sampling/sampling_logp_difference/max": 2.1058051586151123, |
| "sampling/sampling_logp_difference/mean": 0.015573482029139996, |
| "step": 41, |
| "step_time": 25.87153632636182 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009179405751638114, |
| "clip_ratio/high_mean": 0.0009179405751638114, |
| "clip_ratio/low_mean": 2.9995157092344016e-05, |
| "clip_ratio/low_min": 2.9995157092344016e-05, |
| "clip_ratio/region_mean": 0.0009479357162490487, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1955.0, |
| "completions/mean_length": 1204.1199951171875, |
| "completions/mean_terminated_length": 1110.3555908203125, |
| "completions/min_length": 533.0, |
| "completions/min_terminated_length": 533.0, |
| "entropy": 0.18275730907917023, |
| "epoch": 0.11413043478260869, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020292744040489197, |
| "learning_rate": 5e-05, |
| "loss": 0.06081768125295639, |
| "num_tokens": 3425788.0, |
| "reward": 0.3120507597923279, |
| "reward_std": 0.4910523295402527, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304574966430664, |
| "rewards/length_penalty/mean": -0.5879492163658142, |
| "rewards/length_penalty/std": 0.2553500533103943, |
| "sampling/importance_sampling_ratio/max": 2.379917621612549, |
| "sampling/importance_sampling_ratio/mean": 0.9933104515075684, |
| "sampling/importance_sampling_ratio/min": 0.13053026795387268, |
| "sampling/sampling_logp_difference/max": 2.0361502170562744, |
| "sampling/sampling_logp_difference/mean": 0.014663842506706715, |
| "step": 42, |
| "step_time": 25.32822946109809 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008626889786683023, |
| "clip_ratio/high_mean": 0.0008626889786683023, |
| "clip_ratio/low_mean": 0.0001521722035249695, |
| "clip_ratio/low_min": 0.0001521722035249695, |
| "clip_ratio/region_mean": 0.0010148611851036548, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1564.0, |
| "completions/max_terminated_length": 1564.0, |
| "completions/mean_length": 926.1199951171875, |
| "completions/mean_terminated_length": 926.1199951171875, |
| "completions/min_length": 515.0, |
| "completions/min_terminated_length": 515.0, |
| "entropy": 0.21041046380996703, |
| "epoch": 0.11684782608695653, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01616128534078598, |
| "learning_rate": 5e-05, |
| "loss": 0.012639970518648624, |
| "num_tokens": 3474064.0, |
| "reward": 0.5077929496765137, |
| "reward_std": 0.23556646704673767, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.4522070288658142, |
| "rewards/length_penalty/std": 0.1417018473148346, |
| "sampling/importance_sampling_ratio/max": 2.0490548610687256, |
| "sampling/importance_sampling_ratio/mean": 0.9922456741333008, |
| "sampling/importance_sampling_ratio/min": 0.13177694380283356, |
| "sampling/sampling_logp_difference/max": 2.026644706726074, |
| "sampling/sampling_logp_difference/mean": 0.017423400655388832, |
| "step": 43, |
| "step_time": 18.36012065806426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008153841365128756, |
| "clip_ratio/high_mean": 0.0008153841365128756, |
| "clip_ratio/low_mean": 7.955074543133377e-05, |
| "clip_ratio/low_min": 7.955074543133377e-05, |
| "clip_ratio/region_mean": 0.0008949348819442093, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1979.0, |
| "completions/mean_length": 1241.5399169921875, |
| "completions/mean_terminated_length": 1151.933349609375, |
| "completions/min_length": 586.0, |
| "completions/min_terminated_length": 586.0, |
| "entropy": 0.19399587512016297, |
| "epoch": 0.11956521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0202987901866436, |
| "learning_rate": 5e-05, |
| "loss": 0.0733061358332634, |
| "num_tokens": 3539141.0, |
| "reward": 0.3337792754173279, |
| "reward_std": 0.3902702331542969, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.6062207221984863, |
| "rewards/length_penalty/std": 0.21573586761951447, |
| "sampling/importance_sampling_ratio/max": 2.6263153553009033, |
| "sampling/importance_sampling_ratio/mean": 0.992915153503418, |
| "sampling/importance_sampling_ratio/min": 0.17147691547870636, |
| "sampling/sampling_logp_difference/max": 1.7633066177368164, |
| "sampling/sampling_logp_difference/mean": 0.015540734864771366, |
| "step": 44, |
| "step_time": 24.806177518097684 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005664891010383144, |
| "clip_ratio/high_mean": 0.0005664891010383144, |
| "clip_ratio/low_mean": 6.655424949713052e-05, |
| "clip_ratio/low_min": 6.655424949713052e-05, |
| "clip_ratio/region_mean": 0.000633043356356211, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2004.0, |
| "completions/mean_length": 1169.760009765625, |
| "completions/mean_terminated_length": 976.9755859375, |
| "completions/min_length": 465.0, |
| "completions/min_terminated_length": 465.0, |
| "entropy": 0.21446987390518188, |
| "epoch": 0.12228260869565218, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018260451033711433, |
| "learning_rate": 5e-05, |
| "loss": 0.04961071535944939, |
| "num_tokens": 3600569.0, |
| "reward": 0.06882812082767487, |
| "reward_std": 0.7189474701881409, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.5711718797683716, |
| "rewards/length_penalty/std": 0.27816638350486755, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9925439953804016, |
| "sampling/importance_sampling_ratio/min": 0.17214806377887726, |
| "sampling/sampling_logp_difference/max": 1.7594003677368164, |
| "sampling/sampling_logp_difference/mean": 0.016641894355416298, |
| "step": 45, |
| "step_time": 25.038927271263674 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003081215007114224, |
| "clip_ratio/high_mean": 0.0003081215007114224, |
| "clip_ratio/low_mean": 0.00016366775380447507, |
| "clip_ratio/low_min": 0.00016366775380447507, |
| "clip_ratio/region_mean": 0.0004717892617918551, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1565.0, |
| "completions/mean_length": 1187.9599609375, |
| "completions/mean_terminated_length": 972.9500122070312, |
| "completions/min_length": 514.0, |
| "completions/min_terminated_length": 514.0, |
| "entropy": 0.2286642700433731, |
| "epoch": 0.125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015483318828046322, |
| "learning_rate": 5e-05, |
| "loss": 0.04977858066558838, |
| "num_tokens": 3663477.0, |
| "reward": 0.2199414074420929, |
| "reward_std": 0.6239028573036194, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5800585746765137, |
| "rewards/length_penalty/std": 0.23363400995731354, |
| "sampling/importance_sampling_ratio/max": 2.288219451904297, |
| "sampling/importance_sampling_ratio/mean": 0.9922953248023987, |
| "sampling/importance_sampling_ratio/min": 0.13232484459877014, |
| "sampling/sampling_logp_difference/max": 2.0224955081939697, |
| "sampling/sampling_logp_difference/mean": 0.017075827345252037, |
| "step": 46, |
| "step_time": 24.756445694947615 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008448389708064497, |
| "clip_ratio/high_mean": 0.0008448389708064497, |
| "clip_ratio/low_mean": 5.6014435540419075e-05, |
| "clip_ratio/low_min": 5.6014435540419075e-05, |
| "clip_ratio/region_mean": 0.0009008534019812941, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1848.0, |
| "completions/mean_length": 1029.199951171875, |
| "completions/mean_terminated_length": 916.0, |
| "completions/min_length": 646.0, |
| "completions/min_terminated_length": 646.0, |
| "entropy": 0.20336539149284363, |
| "epoch": 0.12771739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02149013802409172, |
| "learning_rate": 5e-05, |
| "loss": 0.051801469177007675, |
| "num_tokens": 3717327.0, |
| "reward": 0.3974609375, |
| "reward_std": 0.49291425943374634, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304577946662903, |
| "rewards/length_penalty/mean": -0.5025390386581421, |
| "rewards/length_penalty/std": 0.22272205352783203, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9924002885818481, |
| "sampling/importance_sampling_ratio/min": 0.12593036890029907, |
| "sampling/sampling_logp_difference/max": 2.072026252746582, |
| "sampling/sampling_logp_difference/mean": 0.016480261459946632, |
| "step": 47, |
| "step_time": 23.89438706357032 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008008693461306393, |
| "clip_ratio/high_mean": 0.0008008693461306393, |
| "clip_ratio/low_mean": 8.54717960464768e-05, |
| "clip_ratio/low_min": 8.54717960464768e-05, |
| "clip_ratio/region_mean": 0.0008863411494530737, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2031.0, |
| "completions/mean_length": 1147.1400146484375, |
| "completions/mean_terminated_length": 1068.8043212890625, |
| "completions/min_length": 510.0, |
| "completions/min_terminated_length": 510.0, |
| "entropy": 0.18503492772579194, |
| "epoch": 0.13043478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020647207275032997, |
| "learning_rate": 5e-05, |
| "loss": 0.03295092657208443, |
| "num_tokens": 3777644.0, |
| "reward": 0.119873046875, |
| "reward_std": 0.5969513654708862, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.5601269602775574, |
| "rewards/length_penalty/std": 0.22607028484344482, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9933799505233765, |
| "sampling/importance_sampling_ratio/min": 0.10150978714227676, |
| "sampling/sampling_logp_difference/max": 2.287600040435791, |
| "sampling/sampling_logp_difference/mean": 0.015032045543193817, |
| "step": 48, |
| "step_time": 24.908888198668137 |
| }, |
| { |
| "clip_ratio/high_max": 0.00028632243920583277, |
| "clip_ratio/high_mean": 0.00028632243920583277, |
| "clip_ratio/low_mean": 0.00013226994196884333, |
| "clip_ratio/low_min": 0.00013226994196884333, |
| "clip_ratio/region_mean": 0.000418592372443527, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1465.0, |
| "completions/mean_length": 1116.2999267578125, |
| "completions/mean_terminated_length": 883.375, |
| "completions/min_length": 406.0, |
| "completions/min_terminated_length": 406.0, |
| "entropy": 0.25495615899562835, |
| "epoch": 0.1331521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015466230921447277, |
| "learning_rate": 5e-05, |
| "loss": 0.05727836489677429, |
| "num_tokens": 3836149.0, |
| "reward": 0.25493162870407104, |
| "reward_std": 0.6429687738418579, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.5450683832168579, |
| "rewards/length_penalty/std": 0.2538842260837555, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9911329746246338, |
| "sampling/importance_sampling_ratio/min": 0.20181740820407867, |
| "sampling/sampling_logp_difference/max": 1.6003918647766113, |
| "sampling/sampling_logp_difference/mean": 0.019320763647556305, |
| "step": 49, |
| "step_time": 24.262222968041897 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003707753901835531, |
| "clip_ratio/high_mean": 0.0003707753901835531, |
| "clip_ratio/low_mean": 0.00015672716835979372, |
| "clip_ratio/low_min": 0.00015672716835979372, |
| "clip_ratio/region_mean": 0.0005275025614537299, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1209.0, |
| "completions/mean_length": 1030.93994140625, |
| "completions/mean_terminated_length": 776.6749877929688, |
| "completions/min_length": 453.0, |
| "completions/min_terminated_length": 453.0, |
| "entropy": 0.2657967835664749, |
| "epoch": 0.1358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01355184055864811, |
| "learning_rate": 5e-05, |
| "loss": 0.03985985368490219, |
| "num_tokens": 3890106.0, |
| "reward": 0.09661132842302322, |
| "reward_std": 0.6983687281608582, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.5033886432647705, |
| "rewards/length_penalty/std": 0.2667604684829712, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9903263449668884, |
| "sampling/importance_sampling_ratio/min": 0.11322663724422455, |
| "sampling/sampling_logp_difference/max": 2.178363800048828, |
| "sampling/sampling_logp_difference/mean": 0.020678939297795296, |
| "step": 50, |
| "step_time": 24.00187555910088 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009819252270972356, |
| "clip_ratio/high_mean": 0.0009819252270972356, |
| "clip_ratio/low_mean": 0.00010533818567637354, |
| "clip_ratio/low_min": 0.00010533818567637354, |
| "clip_ratio/region_mean": 0.0010872634244151413, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1951.0, |
| "completions/max_terminated_length": 1951.0, |
| "completions/mean_length": 946.7799682617188, |
| "completions/mean_terminated_length": 946.7799682617188, |
| "completions/min_length": 465.0, |
| "completions/min_terminated_length": 465.0, |
| "entropy": 0.18979994654655458, |
| "epoch": 0.13858695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013847259804606438, |
| "learning_rate": 5e-05, |
| "loss": 0.019620615988969803, |
| "num_tokens": 3939955.0, |
| "reward": 0.5177050828933716, |
| "reward_std": 0.2139793336391449, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.46229493618011475, |
| "rewards/length_penalty/std": 0.15863463282585144, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9936782121658325, |
| "sampling/importance_sampling_ratio/min": 0.13526712357997894, |
| "sampling/sampling_logp_difference/max": 2.0005037784576416, |
| "sampling/sampling_logp_difference/mean": 0.016131620854139328, |
| "step": 51, |
| "step_time": 22.89465944794938 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009657925576902926, |
| "clip_ratio/high_mean": 0.0009657925576902926, |
| "clip_ratio/low_mean": 3.802108112722635e-05, |
| "clip_ratio/low_min": 3.802108112722635e-05, |
| "clip_ratio/region_mean": 0.001003813638817519, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2013.0, |
| "completions/mean_length": 944.0799560546875, |
| "completions/mean_terminated_length": 873.6170043945312, |
| "completions/min_length": 574.0, |
| "completions/min_terminated_length": 574.0, |
| "entropy": 0.1833059459924698, |
| "epoch": 0.14130434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02072123810648918, |
| "learning_rate": 5e-05, |
| "loss": 0.07086414098739624, |
| "num_tokens": 3989419.0, |
| "reward": 0.2790234386920929, |
| "reward_std": 0.525171160697937, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.46097657084465027, |
| "rewards/length_penalty/std": 0.19479534029960632, |
| "sampling/importance_sampling_ratio/max": 2.803443670272827, |
| "sampling/importance_sampling_ratio/mean": 0.9933574795722961, |
| "sampling/importance_sampling_ratio/min": 0.11414773017168045, |
| "sampling/sampling_logp_difference/max": 2.170261859893799, |
| "sampling/sampling_logp_difference/mean": 0.016782665625214577, |
| "step": 52, |
| "step_time": 23.39199173497036 |
| }, |
| { |
| "clip_ratio/high_max": 0.000828828796511516, |
| "clip_ratio/high_mean": 0.000828828796511516, |
| "clip_ratio/low_mean": 8.272024570032954e-05, |
| "clip_ratio/low_min": 8.272024570032954e-05, |
| "clip_ratio/region_mean": 0.0009115490247495472, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1891.0, |
| "completions/mean_length": 1155.1600341796875, |
| "completions/mean_terminated_length": 931.9500122070312, |
| "completions/min_length": 548.0, |
| "completions/min_terminated_length": 548.0, |
| "entropy": 0.22823919355869293, |
| "epoch": 0.14402173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01979493908584118, |
| "learning_rate": 5e-05, |
| "loss": 0.06918467581272125, |
| "num_tokens": 4049897.0, |
| "reward": 0.05595703050494194, |
| "reward_std": 0.702965497970581, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.5640429854393005, |
| "rewards/length_penalty/std": 0.267347514629364, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9916309118270874, |
| "sampling/importance_sampling_ratio/min": 0.14126569032669067, |
| "sampling/sampling_logp_difference/max": 1.9571127891540527, |
| "sampling/sampling_logp_difference/mean": 0.018662691116333008, |
| "step": 53, |
| "step_time": 24.50819701468572 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008616395702119916, |
| "clip_ratio/high_mean": 0.0008616395702119916, |
| "clip_ratio/low_mean": 0.00011765638773795218, |
| "clip_ratio/low_min": 0.00011765638773795218, |
| "clip_ratio/region_mean": 0.000979295966681093, |
| "completions/clipped_ratio": 0.2800000011920929, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1541.0, |
| "completions/mean_length": 1318.739990234375, |
| "completions/mean_terminated_length": 1035.138916015625, |
| "completions/min_length": 669.0, |
| "completions/min_terminated_length": 669.0, |
| "entropy": 0.23285441994667053, |
| "epoch": 0.14673913043478262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020461982116103172, |
| "learning_rate": 5e-05, |
| "loss": 0.055943913757801056, |
| "num_tokens": 4119654.0, |
| "reward": 0.056083984673023224, |
| "reward_std": 0.6640627980232239, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.6439160108566284, |
| "rewards/length_penalty/std": 0.25468677282333374, |
| "sampling/importance_sampling_ratio/max": 2.6543564796447754, |
| "sampling/importance_sampling_ratio/mean": 0.9915522336959839, |
| "sampling/importance_sampling_ratio/min": 0.1799910068511963, |
| "sampling/sampling_logp_difference/max": 1.7148483991622925, |
| "sampling/sampling_logp_difference/mean": 0.017954660579562187, |
| "step": 54, |
| "step_time": 25.960911376634613 |
| }, |
| { |
| "clip_ratio/high_max": 0.000538797935587354, |
| "clip_ratio/high_mean": 0.000538797935587354, |
| "clip_ratio/low_mean": 0.00018341591639909894, |
| "clip_ratio/low_min": 0.00018341591639909894, |
| "clip_ratio/region_mean": 0.0007222138461656869, |
| "completions/clipped_ratio": 0.35999998450279236, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2046.0, |
| "completions/mean_length": 1324.6199951171875, |
| "completions/mean_terminated_length": 917.71875, |
| "completions/min_length": 520.0, |
| "completions/min_terminated_length": 520.0, |
| "entropy": 0.2338513106107712, |
| "epoch": 0.14945652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023429909721016884, |
| "learning_rate": 5e-05, |
| "loss": 0.011750221252441406, |
| "num_tokens": 4188775.0, |
| "reward": 0.013212890364229679, |
| "reward_std": 0.7507293224334717, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.6467871069908142, |
| "rewards/length_penalty/std": 0.29921308159828186, |
| "sampling/importance_sampling_ratio/max": 2.420614242553711, |
| "sampling/importance_sampling_ratio/mean": 0.9917062520980835, |
| "sampling/importance_sampling_ratio/min": 0.2578577697277069, |
| "sampling/sampling_logp_difference/max": 1.3553471565246582, |
| "sampling/sampling_logp_difference/mean": 0.018062882125377655, |
| "step": 55, |
| "step_time": 25.31786160217598 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010393648175522686, |
| "clip_ratio/high_mean": 0.0010393648175522686, |
| "clip_ratio/low_mean": 9.704843978397548e-05, |
| "clip_ratio/low_min": 9.704843978397548e-05, |
| "clip_ratio/region_mean": 0.001136413251515478, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1945.0, |
| "completions/mean_length": 1084.239990234375, |
| "completions/mean_terminated_length": 779.8947143554688, |
| "completions/min_length": 404.0, |
| "completions/min_terminated_length": 404.0, |
| "entropy": 0.21085838377475738, |
| "epoch": 0.15217391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02038032002747059, |
| "learning_rate": 5e-05, |
| "loss": 0.06151273101568222, |
| "num_tokens": 4246057.0, |
| "reward": 0.030585937201976776, |
| "reward_std": 0.6887958645820618, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.5294140577316284, |
| "rewards/length_penalty/std": 0.3117046356201172, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927169680595398, |
| "sampling/importance_sampling_ratio/min": 0.10817909985780716, |
| "sampling/sampling_logp_difference/max": 2.2239670753479004, |
| "sampling/sampling_logp_difference/mean": 0.01705283299088478, |
| "step": 56, |
| "step_time": 24.31647903425619 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014087805058807135, |
| "clip_ratio/high_mean": 0.0014087805058807135, |
| "clip_ratio/low_mean": 5.1992577209603044e-05, |
| "clip_ratio/low_min": 5.1992577209603044e-05, |
| "clip_ratio/region_mean": 0.0014607730787247418, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1945.0, |
| "completions/mean_length": 1240.3399658203125, |
| "completions/mean_terminated_length": 985.2894897460938, |
| "completions/min_length": 426.0, |
| "completions/min_terminated_length": 426.0, |
| "entropy": 0.23817613422870637, |
| "epoch": 0.15489130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017636116594076157, |
| "learning_rate": 5e-05, |
| "loss": 0.021472664549946785, |
| "num_tokens": 4311314.0, |
| "reward": 0.15436522662639618, |
| "reward_std": 0.684298574924469, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.6056347489356995, |
| "rewards/length_penalty/std": 0.29830121994018555, |
| "sampling/importance_sampling_ratio/max": 2.3371756076812744, |
| "sampling/importance_sampling_ratio/mean": 0.9912160038948059, |
| "sampling/importance_sampling_ratio/min": 0.13857732713222504, |
| "sampling/sampling_logp_difference/max": 1.976326823234558, |
| "sampling/sampling_logp_difference/mean": 0.017901917919516563, |
| "step": 57, |
| "step_time": 25.535622417461127 |
| }, |
| { |
| "clip_ratio/high_max": 0.000502024800516665, |
| "clip_ratio/high_mean": 0.000502024800516665, |
| "clip_ratio/low_mean": 0.0001297363225603476, |
| "clip_ratio/low_min": 0.0001297363225603476, |
| "clip_ratio/region_mean": 0.0006317611201666296, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1942.0, |
| "completions/mean_length": 1429.5599365234375, |
| "completions/mean_terminated_length": 1110.9697265625, |
| "completions/min_length": 603.0, |
| "completions/min_terminated_length": 603.0, |
| "entropy": 0.2496679574251175, |
| "epoch": 0.15760869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02067200466990471, |
| "learning_rate": 5e-05, |
| "loss": 0.03106602281332016, |
| "num_tokens": 4386602.0, |
| "reward": -0.03802734240889549, |
| "reward_std": 0.7067247033119202, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.6980273723602295, |
| "rewards/length_penalty/std": 0.29261118173599243, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9912770390510559, |
| "sampling/importance_sampling_ratio/min": 0.1295800656080246, |
| "sampling/sampling_logp_difference/max": 2.0434563159942627, |
| "sampling/sampling_logp_difference/mean": 0.018928788602352142, |
| "step": 58, |
| "step_time": 26.226440962404013 |
| }, |
| { |
| "clip_ratio/high_max": 0.00048508874606341125, |
| "clip_ratio/high_mean": 0.00048508874606341125, |
| "clip_ratio/low_mean": 0.00011669241503113881, |
| "clip_ratio/low_min": 0.00011669241503113881, |
| "clip_ratio/region_mean": 0.0006017811654601246, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1778.0, |
| "completions/mean_length": 1074.43994140625, |
| "completions/mean_terminated_length": 889.0, |
| "completions/min_length": 464.0, |
| "completions/min_terminated_length": 464.0, |
| "entropy": 0.19868495166301728, |
| "epoch": 0.16032608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018564118072390556, |
| "learning_rate": 5e-05, |
| "loss": 0.06160397827625275, |
| "num_tokens": 4443034.0, |
| "reward": 0.3153710961341858, |
| "reward_std": 0.5987673997879028, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.5246288776397705, |
| "rewards/length_penalty/std": 0.25720861554145813, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929052591323853, |
| "sampling/importance_sampling_ratio/min": 0.1931753307580948, |
| "sampling/sampling_logp_difference/max": 1.6441570520401, |
| "sampling/sampling_logp_difference/mean": 0.017815718427300453, |
| "step": 59, |
| "step_time": 24.022714767139405 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008296219864860177, |
| "clip_ratio/high_mean": 0.0008296219864860177, |
| "clip_ratio/low_mean": 4.176463553449139e-05, |
| "clip_ratio/low_min": 4.176463553449139e-05, |
| "clip_ratio/region_mean": 0.0008713866118341684, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1697.0, |
| "completions/max_terminated_length": 1697.0, |
| "completions/mean_length": 1019.1799926757812, |
| "completions/mean_terminated_length": 1019.1799926757812, |
| "completions/min_length": 509.0, |
| "completions/min_terminated_length": 509.0, |
| "entropy": 0.1617630571126938, |
| "epoch": 0.16304347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016071435064077377, |
| "learning_rate": 5e-05, |
| "loss": 0.00976613350212574, |
| "num_tokens": 4496803.0, |
| "reward": -0.1176464781165123, |
| "reward_std": 0.5020557641983032, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.4976464807987213, |
| "rewards/length_penalty/std": 0.14659054577350616, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9941996932029724, |
| "sampling/importance_sampling_ratio/min": 0.05360739305615425, |
| "sampling/sampling_logp_difference/max": 2.9260683059692383, |
| "sampling/sampling_logp_difference/mean": 0.015030620619654655, |
| "step": 60, |
| "step_time": 20.949455266119912 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008179224882042035, |
| "clip_ratio/high_mean": 0.0008179224882042035, |
| "clip_ratio/low_mean": 4.2844901327043774e-05, |
| "clip_ratio/low_min": 4.2844901327043774e-05, |
| "clip_ratio/region_mean": 0.0008607673895312473, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1610.0, |
| "completions/max_terminated_length": 1610.0, |
| "completions/mean_length": 887.0999755859375, |
| "completions/mean_terminated_length": 887.0999755859375, |
| "completions/min_length": 516.0, |
| "completions/min_terminated_length": 516.0, |
| "entropy": 0.165676087141037, |
| "epoch": 0.16576086956521738, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01667884550988674, |
| "learning_rate": 5e-05, |
| "loss": 0.06641149520874023, |
| "num_tokens": 4543568.0, |
| "reward": 0.566845715045929, |
| "reward_std": 0.10660920292139053, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.43315428495407104, |
| "rewards/length_penalty/std": 0.10660921037197113, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9939702153205872, |
| "sampling/importance_sampling_ratio/min": 0.05033336207270622, |
| "sampling/sampling_logp_difference/max": 2.9890871047973633, |
| "sampling/sampling_logp_difference/mean": 0.016447823494672775, |
| "step": 61, |
| "step_time": 18.75198993459344 |
| }, |
| { |
| "clip_ratio/high_max": 0.000783205998595804, |
| "clip_ratio/high_mean": 0.000783205998595804, |
| "clip_ratio/low_mean": 6.745581340510398e-05, |
| "clip_ratio/low_min": 6.745581340510398e-05, |
| "clip_ratio/region_mean": 0.0008506618207320571, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2024.0, |
| "completions/mean_length": 1215.179931640625, |
| "completions/mean_terminated_length": 1162.021240234375, |
| "completions/min_length": 451.0, |
| "completions/min_terminated_length": 451.0, |
| "entropy": 0.18042704164981843, |
| "epoch": 0.16847826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02352951467037201, |
| "learning_rate": 5e-05, |
| "loss": 0.07232114672660828, |
| "num_tokens": 4607947.0, |
| "reward": 0.24665038287639618, |
| "reward_std": 0.5371196269989014, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.5933496356010437, |
| "rewards/length_penalty/std": 0.2285003960132599, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9935207962989807, |
| "sampling/importance_sampling_ratio/min": 0.13135863840579987, |
| "sampling/sampling_logp_difference/max": 2.0298240184783936, |
| "sampling/sampling_logp_difference/mean": 0.015437403693795204, |
| "step": 62, |
| "step_time": 24.743710697162896 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005814564938191324, |
| "clip_ratio/high_mean": 0.0005814564938191324, |
| "clip_ratio/low_mean": 9.333933558082208e-05, |
| "clip_ratio/low_min": 9.333933558082208e-05, |
| "clip_ratio/region_mean": 0.0006747958424966783, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1855.0, |
| "completions/mean_length": 922.2799682617188, |
| "completions/mean_terminated_length": 899.3060913085938, |
| "completions/min_length": 431.0, |
| "completions/min_terminated_length": 431.0, |
| "entropy": 0.17434330582618712, |
| "epoch": 0.17119565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015940271317958832, |
| "learning_rate": 5e-05, |
| "loss": 0.05170596390962601, |
| "num_tokens": 4657311.0, |
| "reward": 0.3496679663658142, |
| "reward_std": 0.5943698883056641, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.45033204555511475, |
| "rewards/length_penalty/std": 0.2124529331922531, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9934318661689758, |
| "sampling/importance_sampling_ratio/min": 0.10764662176370621, |
| "sampling/sampling_logp_difference/max": 2.2289013862609863, |
| "sampling/sampling_logp_difference/mean": 0.016566412523388863, |
| "step": 63, |
| "step_time": 24.00351930479519 |
| }, |
| { |
| "clip_ratio/high_max": 0.000742131593869999, |
| "clip_ratio/high_mean": 0.000742131593869999, |
| "clip_ratio/low_mean": 3.736222570296377e-05, |
| "clip_ratio/low_min": 3.736222570296377e-05, |
| "clip_ratio/region_mean": 0.000779493828304112, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1313.0, |
| "completions/max_terminated_length": 1313.0, |
| "completions/mean_length": 943.739990234375, |
| "completions/mean_terminated_length": 943.739990234375, |
| "completions/min_length": 565.0, |
| "completions/min_terminated_length": 565.0, |
| "entropy": 0.1540117383003235, |
| "epoch": 0.17391304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016459595412015915, |
| "learning_rate": 5e-05, |
| "loss": 0.01692662388086319, |
| "num_tokens": 4708018.0, |
| "reward": 0.49918943643569946, |
| "reward_std": 0.23669417202472687, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.4608105421066284, |
| "rewards/length_penalty/std": 0.09977121651172638, |
| "sampling/importance_sampling_ratio/max": 2.753690719604492, |
| "sampling/importance_sampling_ratio/mean": 0.9945021271705627, |
| "sampling/importance_sampling_ratio/min": 0.1170341819524765, |
| "sampling/sampling_logp_difference/max": 2.145289182662964, |
| "sampling/sampling_logp_difference/mean": 0.015306901186704636, |
| "step": 64, |
| "step_time": 16.177241111872718 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008999114215839654, |
| "clip_ratio/high_mean": 0.0008999114215839654, |
| "clip_ratio/low_mean": 5.576828407356515e-05, |
| "clip_ratio/low_min": 5.576828407356515e-05, |
| "clip_ratio/region_mean": 0.0009556796809192747, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1970.0, |
| "completions/mean_length": 1085.0599365234375, |
| "completions/mean_terminated_length": 1001.3261108398438, |
| "completions/min_length": 466.0, |
| "completions/min_terminated_length": 466.0, |
| "entropy": 0.18987051248550416, |
| "epoch": 0.1766304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019074536859989166, |
| "learning_rate": 5e-05, |
| "loss": 0.056399617344141006, |
| "num_tokens": 4765981.0, |
| "reward": 0.370185524225235, |
| "reward_std": 0.47647005319595337, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304574966430664, |
| "rewards/length_penalty/mean": -0.5298144817352295, |
| "rewards/length_penalty/std": 0.21904321014881134, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9930379986763, |
| "sampling/importance_sampling_ratio/min": 0.10515164583921432, |
| "sampling/sampling_logp_difference/max": 2.252351760864258, |
| "sampling/sampling_logp_difference/mean": 0.017376262694597244, |
| "step": 65, |
| "step_time": 24.2665373836644 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005626951169688255, |
| "clip_ratio/high_mean": 0.0005626951169688255, |
| "clip_ratio/low_mean": 0.0001472329895477742, |
| "clip_ratio/low_min": 0.0001472329895477742, |
| "clip_ratio/region_mean": 0.0007099281065165997, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1060.0, |
| "completions/mean_length": 951.0599975585938, |
| "completions/mean_terminated_length": 676.8250122070312, |
| "completions/min_length": 395.0, |
| "completions/min_terminated_length": 395.0, |
| "entropy": 0.27960823476314545, |
| "epoch": 0.1793478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016889024525880814, |
| "learning_rate": 5e-05, |
| "loss": 0.004933126736432314, |
| "num_tokens": 4816584.0, |
| "reward": 0.3156152367591858, |
| "reward_std": 0.6881474852561951, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.4643847644329071, |
| "rewards/length_penalty/std": 0.2835237681865692, |
| "sampling/importance_sampling_ratio/max": 2.849069833755493, |
| "sampling/importance_sampling_ratio/mean": 0.9890283346176147, |
| "sampling/importance_sampling_ratio/min": 0.2672819495201111, |
| "sampling/sampling_logp_difference/max": 1.319451093673706, |
| "sampling/sampling_logp_difference/mean": 0.022995835170149803, |
| "step": 66, |
| "step_time": 24.259586106287315 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006742644647601992, |
| "clip_ratio/high_mean": 0.0006742644647601992, |
| "clip_ratio/low_mean": 1.960784284165129e-05, |
| "clip_ratio/low_min": 1.960784284165129e-05, |
| "clip_ratio/region_mean": 0.0006938723148778081, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1684.0, |
| "completions/max_terminated_length": 1684.0, |
| "completions/mean_length": 965.4199829101562, |
| "completions/mean_terminated_length": 965.4199829101562, |
| "completions/min_length": 309.0, |
| "completions/min_terminated_length": 309.0, |
| "entropy": 0.1367795065045357, |
| "epoch": 0.18206521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016154874116182327, |
| "learning_rate": 5e-05, |
| "loss": 0.018381519243121147, |
| "num_tokens": 4867805.0, |
| "reward": 0.10860351473093033, |
| "reward_std": 0.5658869743347168, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.47139647603034973, |
| "rewards/length_penalty/std": 0.17763486504554749, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9949110746383667, |
| "sampling/importance_sampling_ratio/min": 0.10844112932682037, |
| "sampling/sampling_logp_difference/max": 2.221547842025757, |
| "sampling/sampling_logp_difference/mean": 0.014230979606509209, |
| "step": 67, |
| "step_time": 19.920056225033477 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003366426099091768, |
| "clip_ratio/high_mean": 0.0003366426099091768, |
| "clip_ratio/low_mean": 0.0002303594708791934, |
| "clip_ratio/low_min": 0.0002303594708791934, |
| "clip_ratio/region_mean": 0.0005670020822435617, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1588.0, |
| "completions/max_terminated_length": 1588.0, |
| "completions/mean_length": 952.9400024414062, |
| "completions/mean_terminated_length": 952.9400024414062, |
| "completions/min_length": 565.0, |
| "completions/min_terminated_length": 565.0, |
| "entropy": 0.13651491999626159, |
| "epoch": 0.18478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015167438425123692, |
| "learning_rate": 5e-05, |
| "loss": 0.04094483703374863, |
| "num_tokens": 4919212.0, |
| "reward": 0.33469724655151367, |
| "reward_std": 0.5272140502929688, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.4653027355670929, |
| "rewards/length_penalty/std": 0.13752982020378113, |
| "sampling/importance_sampling_ratio/max": 2.9803194999694824, |
| "sampling/importance_sampling_ratio/mean": 0.9950588345527649, |
| "sampling/importance_sampling_ratio/min": 0.16533003747463226, |
| "sampling/sampling_logp_difference/max": 1.799811601638794, |
| "sampling/sampling_logp_difference/mean": 0.014356318861246109, |
| "step": 68, |
| "step_time": 19.00510890292935 |
| }, |
| { |
| "clip_ratio/high_max": 0.000530748994788155, |
| "clip_ratio/high_mean": 0.000530748994788155, |
| "clip_ratio/low_mean": 0.0001243661594344303, |
| "clip_ratio/low_min": 0.0001243661594344303, |
| "clip_ratio/region_mean": 0.0006551151745952666, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1233.0, |
| "completions/mean_length": 966.5799560546875, |
| "completions/mean_terminated_length": 696.2250366210938, |
| "completions/min_length": 532.0, |
| "completions/min_terminated_length": 532.0, |
| "entropy": 0.22109177708625793, |
| "epoch": 0.1875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016588393598794937, |
| "learning_rate": 5e-05, |
| "loss": 0.04386017099022865, |
| "num_tokens": 4969921.0, |
| "reward": 0.12803710997104645, |
| "reward_std": 0.6878940463066101, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.47196289896965027, |
| "rewards/length_penalty/std": 0.2723504304885864, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9917886257171631, |
| "sampling/importance_sampling_ratio/min": 0.20734183490276337, |
| "sampling/sampling_logp_difference/max": 1.5733864307403564, |
| "sampling/sampling_logp_difference/mean": 0.01995372399687767, |
| "step": 69, |
| "step_time": 24.301622559083626 |
| }, |
| { |
| "clip_ratio/high_max": 0.00018317701178602875, |
| "clip_ratio/high_mean": 0.00018317701178602875, |
| "clip_ratio/low_mean": 0.00011933873465750367, |
| "clip_ratio/low_min": 0.00011933873465750367, |
| "clip_ratio/region_mean": 0.0003025157406227663, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1975.0, |
| "completions/mean_length": 1191.3399658203125, |
| "completions/mean_terminated_length": 949.7179565429688, |
| "completions/min_length": 434.0, |
| "completions/min_terminated_length": 434.0, |
| "entropy": 0.12730394005775453, |
| "epoch": 0.19021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017310509458184242, |
| "learning_rate": 5e-05, |
| "loss": 0.04158087819814682, |
| "num_tokens": 5032998.0, |
| "reward": -0.3817089796066284, |
| "reward_std": 0.586998701095581, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5817089676856995, |
| "rewards/length_penalty/std": 0.2990609407424927, |
| "sampling/importance_sampling_ratio/max": 2.500462293624878, |
| "sampling/importance_sampling_ratio/mean": 0.9949836134910583, |
| "sampling/importance_sampling_ratio/min": 0.15331153571605682, |
| "sampling/sampling_logp_difference/max": 1.8752832412719727, |
| "sampling/sampling_logp_difference/mean": 0.012980776838958263, |
| "step": 70, |
| "step_time": 24.772941772127524 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007427642005495727, |
| "clip_ratio/high_mean": 0.0007427642005495727, |
| "clip_ratio/low_mean": 0.00017453260952606797, |
| "clip_ratio/low_min": 0.00017453260952606797, |
| "clip_ratio/region_mean": 0.0009172967867925764, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1884.0, |
| "completions/mean_length": 858.6599731445312, |
| "completions/mean_terminated_length": 782.74462890625, |
| "completions/min_length": 420.0, |
| "completions/min_terminated_length": 420.0, |
| "entropy": 0.20952045917510986, |
| "epoch": 0.19293478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017518984153866768, |
| "learning_rate": 5e-05, |
| "loss": 0.02481238543987274, |
| "num_tokens": 5078291.0, |
| "reward": 0.1607324182987213, |
| "reward_std": 0.5998492240905762, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.41926756501197815, |
| "rewards/length_penalty/std": 0.21910609304904938, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9919474124908447, |
| "sampling/importance_sampling_ratio/min": 0.1041993498802185, |
| "sampling/sampling_logp_difference/max": 2.2614493370056152, |
| "sampling/sampling_logp_difference/mean": 0.021902240812778473, |
| "step": 71, |
| "step_time": 23.054498956771567 |
| }, |
| { |
| "clip_ratio/high_max": 0.001006721018347889, |
| "clip_ratio/high_mean": 0.001006721018347889, |
| "clip_ratio/low_mean": 5.3169773309491575e-05, |
| "clip_ratio/low_min": 5.3169773309491575e-05, |
| "clip_ratio/region_mean": 0.001059890806209296, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1731.0, |
| "completions/max_terminated_length": 1731.0, |
| "completions/mean_length": 759.6599731445312, |
| "completions/mean_terminated_length": 759.6599731445312, |
| "completions/min_length": 546.0, |
| "completions/min_terminated_length": 546.0, |
| "entropy": 0.15967611074447632, |
| "epoch": 0.1956521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016210384666919708, |
| "learning_rate": 5e-05, |
| "loss": 0.0481114462018013, |
| "num_tokens": 5119944.0, |
| "reward": 0.4290722608566284, |
| "reward_std": 0.48313963413238525, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.37092772126197815, |
| "rewards/length_penalty/std": 0.09718189388513565, |
| "sampling/importance_sampling_ratio/max": 2.467366933822632, |
| "sampling/importance_sampling_ratio/mean": 0.993756115436554, |
| "sampling/importance_sampling_ratio/min": 0.10246583074331284, |
| "sampling/sampling_logp_difference/max": 2.278225898742676, |
| "sampling/sampling_logp_difference/mean": 0.01932191662490368, |
| "step": 72, |
| "step_time": 19.477228795876727 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013294589822180568, |
| "clip_ratio/high_mean": 0.0013294589822180568, |
| "clip_ratio/low_mean": 3.77274613128975e-05, |
| "clip_ratio/low_min": 3.77274613128975e-05, |
| "clip_ratio/region_mean": 0.0013671864639036358, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1047.0, |
| "completions/mean_length": 1016.7599487304688, |
| "completions/mean_terminated_length": 758.9500122070312, |
| "completions/min_length": 481.0, |
| "completions/min_terminated_length": 481.0, |
| "entropy": 0.20307525396347045, |
| "epoch": 0.1983695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014327866956591606, |
| "learning_rate": 5e-05, |
| "loss": 0.03987980633974075, |
| "num_tokens": 5173192.0, |
| "reward": 0.10353515297174454, |
| "reward_std": 0.6840619444847107, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.4964648485183716, |
| "rewards/length_penalty/std": 0.25960496068000793, |
| "sampling/importance_sampling_ratio/max": 2.748769760131836, |
| "sampling/importance_sampling_ratio/mean": 0.9922918677330017, |
| "sampling/importance_sampling_ratio/min": 0.051035378128290176, |
| "sampling/sampling_logp_difference/max": 2.975236177444458, |
| "sampling/sampling_logp_difference/mean": 0.018849551677703857, |
| "step": 73, |
| "step_time": 24.514374701771885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009000395250041038, |
| "clip_ratio/high_mean": 0.0009000395250041038, |
| "clip_ratio/low_mean": 0.00015482750604860484, |
| "clip_ratio/low_min": 0.00015482750604860484, |
| "clip_ratio/region_mean": 0.0010548670194111764, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1832.0, |
| "completions/mean_length": 884.0, |
| "completions/mean_terminated_length": 809.7020874023438, |
| "completions/min_length": 370.0, |
| "completions/min_terminated_length": 370.0, |
| "entropy": 0.2515583157539368, |
| "epoch": 0.20108695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018280453979969025, |
| "learning_rate": 5e-05, |
| "loss": 0.07817187905311584, |
| "num_tokens": 5220722.0, |
| "reward": -0.03164062276482582, |
| "reward_std": 0.6271845698356628, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.431640625, |
| "rewards/length_penalty/std": 0.22327032685279846, |
| "sampling/importance_sampling_ratio/max": 2.1104793548583984, |
| "sampling/importance_sampling_ratio/mean": 0.9905590415000916, |
| "sampling/importance_sampling_ratio/min": 0.0870686024427414, |
| "sampling/sampling_logp_difference/max": 2.441058874130249, |
| "sampling/sampling_logp_difference/mean": 0.02339356765151024, |
| "step": 74, |
| "step_time": 23.44279443915002 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004406559950439259, |
| "clip_ratio/high_mean": 0.0004406559950439259, |
| "clip_ratio/low_mean": 0.00014615571562899278, |
| "clip_ratio/low_min": 0.00014615571562899278, |
| "clip_ratio/region_mean": 0.000586811697576195, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1602.0, |
| "completions/mean_length": 830.3999633789062, |
| "completions/mean_terminated_length": 779.6666870117188, |
| "completions/min_length": 278.0, |
| "completions/min_terminated_length": 278.0, |
| "entropy": 0.21685439348220825, |
| "epoch": 0.20380434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018299926072359085, |
| "learning_rate": 5e-05, |
| "loss": 0.059524986892938614, |
| "num_tokens": 5264572.0, |
| "reward": 0.43453124165534973, |
| "reward_std": 0.5175331234931946, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.40546876192092896, |
| "rewards/length_penalty/std": 0.22259218990802765, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9922915697097778, |
| "sampling/importance_sampling_ratio/min": 0.08350156992673874, |
| "sampling/sampling_logp_difference/max": 2.4828898906707764, |
| "sampling/sampling_logp_difference/mean": 0.021242547780275345, |
| "step": 75, |
| "step_time": 23.03285480872728 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007463882677257061, |
| "clip_ratio/high_mean": 0.0007463882677257061, |
| "clip_ratio/low_mean": 7.58719674195163e-05, |
| "clip_ratio/low_min": 7.58719674195163e-05, |
| "clip_ratio/region_mean": 0.0008222602307796478, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1854.0, |
| "completions/mean_length": 814.8200073242188, |
| "completions/mean_terminated_length": 614.0697631835938, |
| "completions/min_length": 150.0, |
| "completions/min_terminated_length": 150.0, |
| "entropy": 0.23157910704612733, |
| "epoch": 0.20652173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016244014725089073, |
| "learning_rate": 5e-05, |
| "loss": 0.019556419923901558, |
| "num_tokens": 5308653.0, |
| "reward": 0.3821386694908142, |
| "reward_std": 0.6754101514816284, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.3978613317012787, |
| "rewards/length_penalty/std": 0.28565603494644165, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9915950894355774, |
| "sampling/importance_sampling_ratio/min": 0.20373162627220154, |
| "sampling/sampling_logp_difference/max": 1.9097907543182373, |
| "sampling/sampling_logp_difference/mean": 0.021138090640306473, |
| "step": 76, |
| "step_time": 23.710425530094653 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006072747666621581, |
| "clip_ratio/high_mean": 0.0006072747666621581, |
| "clip_ratio/low_mean": 9.39797842875123e-05, |
| "clip_ratio/low_min": 9.39797842875123e-05, |
| "clip_ratio/region_mean": 0.0007012545596808195, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1628.0, |
| "completions/mean_length": 759.0799560546875, |
| "completions/mean_terminated_length": 676.8084716796875, |
| "completions/min_length": 257.0, |
| "completions/min_terminated_length": 257.0, |
| "entropy": 0.21539296507835387, |
| "epoch": 0.20923913043478262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020625092089176178, |
| "learning_rate": 5e-05, |
| "loss": 0.016749916598200798, |
| "num_tokens": 5349167.0, |
| "reward": 0.4893554449081421, |
| "reward_std": 0.5410571098327637, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.37064453959465027, |
| "rewards/length_penalty/std": 0.2292376607656479, |
| "sampling/importance_sampling_ratio/max": 2.495506525039673, |
| "sampling/importance_sampling_ratio/mean": 0.9919021725654602, |
| "sampling/importance_sampling_ratio/min": 0.16004765033721924, |
| "sampling/sampling_logp_difference/max": 1.8322837352752686, |
| "sampling/sampling_logp_difference/mean": 0.021835042163729668, |
| "step": 77, |
| "step_time": 22.753829170949757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003826598927844316, |
| "clip_ratio/high_mean": 0.0003826598927844316, |
| "clip_ratio/low_mean": 0.00022446984949056059, |
| "clip_ratio/low_min": 0.00022446984949056059, |
| "clip_ratio/region_mean": 0.0006071297393646091, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1993.0, |
| "completions/mean_length": 848.0999755859375, |
| "completions/mean_terminated_length": 798.1041870117188, |
| "completions/min_length": 407.0, |
| "completions/min_terminated_length": 407.0, |
| "entropy": 0.1559257060289383, |
| "epoch": 0.21195652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01592531055212021, |
| "learning_rate": 5e-05, |
| "loss": 0.06760542839765549, |
| "num_tokens": 5393872.0, |
| "reward": 0.18588866293430328, |
| "reward_std": 0.6589588522911072, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.41411131620407104, |
| "rewards/length_penalty/std": 0.2256016880273819, |
| "sampling/importance_sampling_ratio/max": 2.6483066082000732, |
| "sampling/importance_sampling_ratio/mean": 0.9938704967498779, |
| "sampling/importance_sampling_ratio/min": 0.11605721712112427, |
| "sampling/sampling_logp_difference/max": 2.153671979904175, |
| "sampling/sampling_logp_difference/mean": 0.018067849799990654, |
| "step": 78, |
| "step_time": 23.126321063842624 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009475421567913145, |
| "clip_ratio/high_mean": 0.0009475421567913145, |
| "clip_ratio/low_mean": 7.389541424345225e-05, |
| "clip_ratio/low_min": 7.389541424345225e-05, |
| "clip_ratio/region_mean": 0.0010214375972282142, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1342.0, |
| "completions/mean_length": 826.260009765625, |
| "completions/mean_terminated_length": 801.3265380859375, |
| "completions/min_length": 454.0, |
| "completions/min_terminated_length": 454.0, |
| "entropy": 0.14301405251026153, |
| "epoch": 0.21467391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022472739219665527, |
| "learning_rate": 5e-05, |
| "loss": 0.05088501423597336, |
| "num_tokens": 5438645.0, |
| "reward": 0.35655272006988525, |
| "reward_std": 0.44962450861930847, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.4034472703933716, |
| "rewards/length_penalty/std": 0.14060236513614655, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9948509335517883, |
| "sampling/importance_sampling_ratio/min": 0.13412797451019287, |
| "sampling/sampling_logp_difference/max": 2.0089609622955322, |
| "sampling/sampling_logp_difference/mean": 0.01816740445792675, |
| "step": 79, |
| "step_time": 23.127719740383327 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005540229729376733, |
| "clip_ratio/high_mean": 0.0005540229729376733, |
| "clip_ratio/low_mean": 5.406462587416172e-05, |
| "clip_ratio/low_min": 5.406462587416172e-05, |
| "clip_ratio/region_mean": 0.0006080876046326011, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1760.0, |
| "completions/max_terminated_length": 1760.0, |
| "completions/mean_length": 701.0399780273438, |
| "completions/mean_terminated_length": 701.0399780273438, |
| "completions/min_length": 462.0, |
| "completions/min_terminated_length": 462.0, |
| "entropy": 0.1606625497341156, |
| "epoch": 0.21739130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017060762271285057, |
| "learning_rate": 5e-05, |
| "loss": 0.053541149944067, |
| "num_tokens": 5476857.0, |
| "reward": 0.057695310562849045, |
| "reward_std": 0.5497966408729553, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.34230467677116394, |
| "rewards/length_penalty/std": 0.09248920530080795, |
| "sampling/importance_sampling_ratio/max": 2.2936506271362305, |
| "sampling/importance_sampling_ratio/mean": 0.9937153458595276, |
| "sampling/importance_sampling_ratio/min": 0.13374976813793182, |
| "sampling/sampling_logp_difference/max": 2.011784553527832, |
| "sampling/sampling_logp_difference/mean": 0.018699120730161667, |
| "step": 80, |
| "step_time": 19.293394381180406 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008017016400117428, |
| "clip_ratio/high_mean": 0.0008017016400117428, |
| "clip_ratio/low_mean": 0.00011225517664570361, |
| "clip_ratio/low_min": 0.00011225517664570361, |
| "clip_ratio/region_mean": 0.0009139568195678293, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1296.0, |
| "completions/max_terminated_length": 1296.0, |
| "completions/mean_length": 820.6799926757812, |
| "completions/mean_terminated_length": 820.6799926757812, |
| "completions/min_length": 431.0, |
| "completions/min_terminated_length": 431.0, |
| "entropy": 0.15851948261260987, |
| "epoch": 0.22010869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017597660422325134, |
| "learning_rate": 5e-05, |
| "loss": 0.009531941264867783, |
| "num_tokens": 5522351.0, |
| "reward": 0.25927734375, |
| "reward_std": 0.422261506319046, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.4007226526737213, |
| "rewards/length_penalty/std": 0.10731947422027588, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9941749572753906, |
| "sampling/importance_sampling_ratio/min": 0.17774875462055206, |
| "sampling/sampling_logp_difference/max": 1.7273842096328735, |
| "sampling/sampling_logp_difference/mean": 0.018384624272584915, |
| "step": 81, |
| "step_time": 16.074628691887483 |
| }, |
| { |
| "clip_ratio/high_max": 0.001255881180986762, |
| "clip_ratio/high_mean": 0.001255881180986762, |
| "clip_ratio/low_mean": 0.00010265577293466777, |
| "clip_ratio/low_min": 0.00010265577293466777, |
| "clip_ratio/region_mean": 0.0013585369568318128, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 886.0, |
| "completions/max_terminated_length": 886.0, |
| "completions/mean_length": 550.739990234375, |
| "completions/mean_terminated_length": 550.739990234375, |
| "completions/min_length": 311.0, |
| "completions/min_terminated_length": 311.0, |
| "entropy": 0.14662574827671052, |
| "epoch": 0.22282608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014600266702473164, |
| "learning_rate": 5e-05, |
| "loss": 0.009336000308394432, |
| "num_tokens": 5552738.0, |
| "reward": 0.5110839605331421, |
| "reward_std": 0.4766097366809845, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.2689160108566284, |
| "rewards/length_penalty/std": 0.07129478454589844, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.994766354560852, |
| "sampling/importance_sampling_ratio/min": 0.1095326840877533, |
| "sampling/sampling_logp_difference/max": 2.2115323543548584, |
| "sampling/sampling_logp_difference/mean": 0.02184543013572693, |
| "step": 82, |
| "step_time": 10.690508378203958 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005720254499465227, |
| "clip_ratio/high_mean": 0.0005720254499465227, |
| "clip_ratio/low_mean": 4.0096230804920197e-05, |
| "clip_ratio/low_min": 4.0096230804920197e-05, |
| "clip_ratio/region_mean": 0.0006121216807514429, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1133.0, |
| "completions/max_terminated_length": 1133.0, |
| "completions/mean_length": 556.8599853515625, |
| "completions/mean_terminated_length": 556.8599853515625, |
| "completions/min_length": 349.0, |
| "completions/min_terminated_length": 349.0, |
| "entropy": 0.15898979604244232, |
| "epoch": 0.22554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01600862294435501, |
| "learning_rate": 5e-05, |
| "loss": -0.0012336941435933113, |
| "num_tokens": 5583011.0, |
| "reward": 0.2680957019329071, |
| "reward_std": 0.5045760869979858, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.2719042897224426, |
| "rewards/length_penalty/std": 0.0767233818769455, |
| "sampling/importance_sampling_ratio/max": 2.753527879714966, |
| "sampling/importance_sampling_ratio/mean": 0.9942069053649902, |
| "sampling/importance_sampling_ratio/min": 0.08033192902803421, |
| "sampling/sampling_logp_difference/max": 2.521588087081909, |
| "sampling/sampling_logp_difference/mean": 0.021603433415293694, |
| "step": 83, |
| "step_time": 12.889199507189915 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005918114853557199, |
| "clip_ratio/high_mean": 0.0005918114853557199, |
| "clip_ratio/low_mean": 8.836851920932532e-05, |
| "clip_ratio/low_min": 8.836851920932532e-05, |
| "clip_ratio/region_mean": 0.0006801799929235131, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1695.0, |
| "completions/max_terminated_length": 1695.0, |
| "completions/mean_length": 675.8999633789062, |
| "completions/mean_terminated_length": 675.8999633789062, |
| "completions/min_length": 288.0, |
| "completions/min_terminated_length": 288.0, |
| "entropy": 0.14046733379364013, |
| "epoch": 0.22826086956521738, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017347121611237526, |
| "learning_rate": 5e-05, |
| "loss": 0.034702517092227936, |
| "num_tokens": 5619786.0, |
| "reward": 0.469970703125, |
| "reward_std": 0.39906755089759827, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.33002930879592896, |
| "rewards/length_penalty/std": 0.20127812027931213, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9947786927223206, |
| "sampling/importance_sampling_ratio/min": 0.18437546491622925, |
| "sampling/sampling_logp_difference/max": 1.6907809972763062, |
| "sampling/sampling_logp_difference/mean": 0.0183523278683424, |
| "step": 84, |
| "step_time": 19.66053499514237 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010052890516817569, |
| "clip_ratio/high_mean": 0.0010052890516817569, |
| "clip_ratio/low_mean": 0.00017086818406824021, |
| "clip_ratio/low_min": 0.00017086818406824021, |
| "clip_ratio/region_mean": 0.0011761572386603802, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1977.0, |
| "completions/mean_length": 942.2599487304688, |
| "completions/mean_terminated_length": 699.5365600585938, |
| "completions/min_length": 308.0, |
| "completions/min_terminated_length": 308.0, |
| "entropy": 0.27836973667144777, |
| "epoch": 0.23097826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.027903197333216667, |
| "learning_rate": 5e-05, |
| "loss": 0.12671905755996704, |
| "num_tokens": 5668919.0, |
| "reward": 0.13991211354732513, |
| "reward_std": 0.6963438391685486, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.4600878953933716, |
| "rewards/length_penalty/std": 0.3180088400840759, |
| "sampling/importance_sampling_ratio/max": 2.491806745529175, |
| "sampling/importance_sampling_ratio/mean": 0.9892483949661255, |
| "sampling/importance_sampling_ratio/min": 0.11257338523864746, |
| "sampling/sampling_logp_difference/max": 2.184149980545044, |
| "sampling/sampling_logp_difference/mean": 0.025783531367778778, |
| "step": 85, |
| "step_time": 23.433202574960887 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008290928497444838, |
| "clip_ratio/high_mean": 0.0008290928497444838, |
| "clip_ratio/low_mean": 0.000351838962524198, |
| "clip_ratio/low_min": 0.000351838962524198, |
| "clip_ratio/region_mean": 0.0011809318093582988, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 726.0, |
| "completions/max_terminated_length": 726.0, |
| "completions/mean_length": 404.2799987792969, |
| "completions/mean_terminated_length": 404.2799987792969, |
| "completions/min_length": 213.0, |
| "completions/min_terminated_length": 213.0, |
| "entropy": 0.14651564359664918, |
| "epoch": 0.23369565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013358832336962223, |
| "learning_rate": 5e-05, |
| "loss": 0.022671468555927277, |
| "num_tokens": 5691233.0, |
| "reward": 0.8025976419448853, |
| "reward_std": 0.05789410322904587, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.19740234315395355, |
| "rewards/length_penalty/std": 0.057894106954336166, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9948070645332336, |
| "sampling/importance_sampling_ratio/min": 0.3225119411945343, |
| "sampling/sampling_logp_difference/max": 1.4580408334732056, |
| "sampling/sampling_logp_difference/mean": 0.023679958656430244, |
| "step": 86, |
| "step_time": 8.556200941326097 |
| }, |
| { |
| "clip_ratio/high_max": 0.00039820867532398553, |
| "clip_ratio/high_mean": 0.00039820867532398553, |
| "clip_ratio/low_mean": 0.00016748259076848626, |
| "clip_ratio/low_min": 0.00016748259076848626, |
| "clip_ratio/region_mean": 0.0005656912748236209, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1219.0, |
| "completions/max_terminated_length": 1219.0, |
| "completions/mean_length": 626.5799560546875, |
| "completions/mean_terminated_length": 626.5799560546875, |
| "completions/min_length": 209.0, |
| "completions/min_terminated_length": 209.0, |
| "entropy": 0.14620184004306794, |
| "epoch": 0.23641304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014849641360342503, |
| "learning_rate": 5e-05, |
| "loss": 0.05206526815891266, |
| "num_tokens": 5724772.0, |
| "reward": 0.29405272006988525, |
| "reward_std": 0.5525745153427124, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.30594727396965027, |
| "rewards/length_penalty/std": 0.13335204124450684, |
| "sampling/importance_sampling_ratio/max": 2.649275302886963, |
| "sampling/importance_sampling_ratio/mean": 0.9941005706787109, |
| "sampling/importance_sampling_ratio/min": 0.16833527386188507, |
| "sampling/sampling_logp_difference/max": 1.7817976474761963, |
| "sampling/sampling_logp_difference/mean": 0.018479688093066216, |
| "step": 87, |
| "step_time": 14.042173262918368 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006503341835923492, |
| "clip_ratio/high_mean": 0.0006503341835923492, |
| "clip_ratio/low_mean": 8.594756945967674e-05, |
| "clip_ratio/low_min": 8.594756945967674e-05, |
| "clip_ratio/region_mean": 0.000736281753052026, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 894.0, |
| "completions/max_terminated_length": 894.0, |
| "completions/mean_length": 491.5799865722656, |
| "completions/mean_terminated_length": 491.5799865722656, |
| "completions/min_length": 249.0, |
| "completions/min_terminated_length": 249.0, |
| "entropy": 0.1407701775431633, |
| "epoch": 0.2391304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01568138599395752, |
| "learning_rate": 5e-05, |
| "loss": 0.03671112656593323, |
| "num_tokens": 5752241.0, |
| "reward": 0.7599706649780273, |
| "reward_std": 0.08543706685304642, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.24002929031848907, |
| "rewards/length_penalty/std": 0.08543706685304642, |
| "sampling/importance_sampling_ratio/max": 2.9965686798095703, |
| "sampling/importance_sampling_ratio/mean": 0.9948350787162781, |
| "sampling/importance_sampling_ratio/min": 0.17062287032604218, |
| "sampling/sampling_logp_difference/max": 1.7682995796203613, |
| "sampling/sampling_logp_difference/mean": 0.01965334452688694, |
| "step": 88, |
| "step_time": 11.003475670004264 |
| }, |
| { |
| "clip_ratio/high_max": 0.000445453537395224, |
| "clip_ratio/high_mean": 0.000445453537395224, |
| "clip_ratio/low_mean": 0.00010250343621009961, |
| "clip_ratio/low_min": 0.00010250343621009961, |
| "clip_ratio/region_mean": 0.0005479569750605151, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1913.0, |
| "completions/mean_length": 743.2000122070312, |
| "completions/mean_terminated_length": 688.8333740234375, |
| "completions/min_length": 250.0, |
| "completions/min_terminated_length": 250.0, |
| "entropy": 0.1520453631877899, |
| "epoch": 0.2418478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023198934271931648, |
| "learning_rate": 5e-05, |
| "loss": 0.010194681584835052, |
| "num_tokens": 5792861.0, |
| "reward": 0.29710936546325684, |
| "reward_std": 0.6285626888275146, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.3628906309604645, |
| "rewards/length_penalty/std": 0.24505671858787537, |
| "sampling/importance_sampling_ratio/max": 2.8949134349823, |
| "sampling/importance_sampling_ratio/mean": 0.9940477013587952, |
| "sampling/importance_sampling_ratio/min": 0.24097231030464172, |
| "sampling/sampling_logp_difference/max": 1.4230732917785645, |
| "sampling/sampling_logp_difference/mean": 0.017822299152612686, |
| "step": 89, |
| "step_time": 22.883739611133933 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008915362093830481, |
| "clip_ratio/high_mean": 0.0008915362093830481, |
| "clip_ratio/low_mean": 0.00012391345808282495, |
| "clip_ratio/low_min": 0.00012391345808282495, |
| "clip_ratio/region_mean": 0.0010154496791074052, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2047.0, |
| "completions/mean_length": 863.6399536132812, |
| "completions/mean_terminated_length": 814.2916870117188, |
| "completions/min_length": 362.0, |
| "completions/min_terminated_length": 362.0, |
| "entropy": 0.1281779631972313, |
| "epoch": 0.24456521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017280666157603264, |
| "learning_rate": 5e-05, |
| "loss": 0.03663965314626694, |
| "num_tokens": 5839073.0, |
| "reward": 0.4983007609844208, |
| "reward_std": 0.44724196195602417, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.4216992259025574, |
| "rewards/length_penalty/std": 0.26747414469718933, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9952991604804993, |
| "sampling/importance_sampling_ratio/min": 0.16416248679161072, |
| "sampling/sampling_logp_difference/max": 1.806898593902588, |
| "sampling/sampling_logp_difference/mean": 0.01531835924834013, |
| "step": 90, |
| "step_time": 23.317600117065012 |
| }, |
| { |
| "clip_ratio/high_max": 0.000796247145626694, |
| "clip_ratio/high_mean": 0.000796247145626694, |
| "clip_ratio/low_mean": 9.555761935189366e-05, |
| "clip_ratio/low_min": 9.555761935189366e-05, |
| "clip_ratio/region_mean": 0.0008918047766201198, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1763.0, |
| "completions/mean_length": 693.719970703125, |
| "completions/mean_terminated_length": 666.0816040039062, |
| "completions/min_length": 98.0, |
| "completions/min_terminated_length": 98.0, |
| "entropy": 0.15058106780052186, |
| "epoch": 0.24728260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019430285319685936, |
| "learning_rate": 5e-05, |
| "loss": 0.02801748737692833, |
| "num_tokens": 5878929.0, |
| "reward": 0.601269543170929, |
| "reward_std": 0.40769532322883606, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.33873045444488525, |
| "rewards/length_penalty/std": 0.24405017495155334, |
| "sampling/importance_sampling_ratio/max": 2.51088547706604, |
| "sampling/importance_sampling_ratio/mean": 0.994093656539917, |
| "sampling/importance_sampling_ratio/min": 0.2567068040370941, |
| "sampling/sampling_logp_difference/max": 1.3598206043243408, |
| "sampling/sampling_logp_difference/mean": 0.017727866768836975, |
| "step": 91, |
| "step_time": 23.424256341997534 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009729682933539152, |
| "clip_ratio/high_mean": 0.0009729682933539152, |
| "clip_ratio/low_mean": 2.3421946389134972e-05, |
| "clip_ratio/low_min": 2.3421946389134972e-05, |
| "clip_ratio/region_mean": 0.0009963902295567096, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1942.0, |
| "completions/mean_length": 702.3599853515625, |
| "completions/mean_terminated_length": 585.3478393554688, |
| "completions/min_length": 167.0, |
| "completions/min_terminated_length": 167.0, |
| "entropy": 0.1468707501888275, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019020741805434227, |
| "learning_rate": 5e-05, |
| "loss": 0.024482248350977898, |
| "num_tokens": 5916437.0, |
| "reward": 0.3770507872104645, |
| "reward_std": 0.5595439672470093, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.3429492115974426, |
| "rewards/length_penalty/std": 0.2876156270503998, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9945229291915894, |
| "sampling/importance_sampling_ratio/min": 0.12576083838939667, |
| "sampling/sampling_logp_difference/max": 2.073373317718506, |
| "sampling/sampling_logp_difference/mean": 0.018665308132767677, |
| "step": 92, |
| "step_time": 23.23252943577245 |
| }, |
| { |
| "clip_ratio/high_max": 0.000579733302583918, |
| "clip_ratio/high_mean": 0.000579733302583918, |
| "clip_ratio/low_mean": 8.087056921795011e-05, |
| "clip_ratio/low_min": 8.087056921795011e-05, |
| "clip_ratio/region_mean": 0.0006606038776226341, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1916.0, |
| "completions/mean_length": 962.719970703125, |
| "completions/mean_terminated_length": 786.0465087890625, |
| "completions/min_length": 336.0, |
| "completions/min_terminated_length": 336.0, |
| "entropy": 0.19105736017227173, |
| "epoch": 0.25271739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024748774245381355, |
| "learning_rate": 5e-05, |
| "loss": 0.035268157720565796, |
| "num_tokens": 5970193.0, |
| "reward": 0.2699218690395355, |
| "reward_std": 0.6784000992774963, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.47007811069488525, |
| "rewards/length_penalty/std": 0.2857426106929779, |
| "sampling/importance_sampling_ratio/max": 2.238551139831543, |
| "sampling/importance_sampling_ratio/mean": 0.9926732778549194, |
| "sampling/importance_sampling_ratio/min": 0.17437584698200226, |
| "sampling/sampling_logp_difference/max": 1.74654221534729, |
| "sampling/sampling_logp_difference/mean": 0.01878998428583145, |
| "step": 93, |
| "step_time": 24.461774740368128 |
| }, |
| { |
| "clip_ratio/high_max": 0.000649976628483273, |
| "clip_ratio/high_mean": 0.000649976628483273, |
| "clip_ratio/low_mean": 0.00017161076539196074, |
| "clip_ratio/low_min": 0.00017161076539196074, |
| "clip_ratio/region_mean": 0.0008215873909648508, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1606.0, |
| "completions/max_terminated_length": 1606.0, |
| "completions/mean_length": 668.9199829101562, |
| "completions/mean_terminated_length": 668.9199829101562, |
| "completions/min_length": 306.0, |
| "completions/min_terminated_length": 306.0, |
| "entropy": 0.17675926685333251, |
| "epoch": 0.2554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01949072815477848, |
| "learning_rate": 5e-05, |
| "loss": 0.02970695123076439, |
| "num_tokens": 6009669.0, |
| "reward": 0.1533789038658142, |
| "reward_std": 0.5026692748069763, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.32662108540534973, |
| "rewards/length_penalty/std": 0.16069523990154266, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9936524629592896, |
| "sampling/importance_sampling_ratio/min": 0.17633819580078125, |
| "sampling/sampling_logp_difference/max": 1.7353515625, |
| "sampling/sampling_logp_difference/mean": 0.020932145416736603, |
| "step": 94, |
| "step_time": 18.958991665160283 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010454685078002512, |
| "clip_ratio/high_mean": 0.0010454685078002512, |
| "clip_ratio/low_mean": 0.00018632706487551332, |
| "clip_ratio/low_min": 0.00018632706487551332, |
| "clip_ratio/region_mean": 0.0012317955726757646, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 730.0, |
| "completions/max_terminated_length": 730.0, |
| "completions/mean_length": 438.8800048828125, |
| "completions/mean_terminated_length": 438.8800048828125, |
| "completions/min_length": 216.0, |
| "completions/min_terminated_length": 216.0, |
| "entropy": 0.1665761500597, |
| "epoch": 0.25815217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01713760383427143, |
| "learning_rate": 5e-05, |
| "loss": 0.02406269498169422, |
| "num_tokens": 6033983.0, |
| "reward": 0.7857031226158142, |
| "reward_std": 0.06166134029626846, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.2142968773841858, |
| "rewards/length_penalty/std": 0.06166134402155876, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9940052628517151, |
| "sampling/importance_sampling_ratio/min": 0.10517227649688721, |
| "sampling/sampling_logp_difference/max": 2.2521555423736572, |
| "sampling/sampling_logp_difference/mean": 0.023947253823280334, |
| "step": 95, |
| "step_time": 8.727350569097325 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009204381611198186, |
| "clip_ratio/high_mean": 0.0009204381611198186, |
| "clip_ratio/low_mean": 7.934253371786327e-05, |
| "clip_ratio/low_min": 7.934253371786327e-05, |
| "clip_ratio/region_mean": 0.000999780697748065, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 961.0, |
| "completions/max_terminated_length": 961.0, |
| "completions/mean_length": 521.8599853515625, |
| "completions/mean_terminated_length": 521.8599853515625, |
| "completions/min_length": 197.0, |
| "completions/min_terminated_length": 197.0, |
| "entropy": 0.1356882095336914, |
| "epoch": 0.2608695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02230748161673546, |
| "learning_rate": 5e-05, |
| "loss": 0.033758148550987244, |
| "num_tokens": 6063606.0, |
| "reward": 0.5251855254173279, |
| "reward_std": 0.495756596326828, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.2548144459724426, |
| "rewards/length_penalty/std": 0.09907565265893936, |
| "sampling/importance_sampling_ratio/max": 2.202416181564331, |
| "sampling/importance_sampling_ratio/mean": 0.9945613145828247, |
| "sampling/importance_sampling_ratio/min": 0.20575757324695587, |
| "sampling/sampling_logp_difference/max": 1.5810565948486328, |
| "sampling/sampling_logp_difference/mean": 0.019358227029442787, |
| "step": 96, |
| "step_time": 11.922741995193064 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008645437832456082, |
| "clip_ratio/high_mean": 0.0008645437832456082, |
| "clip_ratio/low_mean": 8.568980265408755e-05, |
| "clip_ratio/low_min": 8.568980265408755e-05, |
| "clip_ratio/region_mean": 0.0009502335858996957, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 951.0, |
| "completions/max_terminated_length": 951.0, |
| "completions/mean_length": 471.3799743652344, |
| "completions/mean_terminated_length": 471.3799743652344, |
| "completions/min_length": 218.0, |
| "completions/min_terminated_length": 218.0, |
| "entropy": 0.16626610457897187, |
| "epoch": 0.26358695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015668677166104317, |
| "learning_rate": 5e-05, |
| "loss": -0.005791757255792618, |
| "num_tokens": 6090105.0, |
| "reward": 0.7498339414596558, |
| "reward_std": 0.17372536659240723, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.2301660180091858, |
| "rewards/length_penalty/std": 0.09313111007213593, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9936684966087341, |
| "sampling/importance_sampling_ratio/min": 0.17704980075359344, |
| "sampling/sampling_logp_difference/max": 1.7313241958618164, |
| "sampling/sampling_logp_difference/mean": 0.02380496822297573, |
| "step": 97, |
| "step_time": 11.000973380170763 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008191536413505674, |
| "clip_ratio/high_mean": 0.0008191536413505674, |
| "clip_ratio/low_mean": 7.509166607633233e-05, |
| "clip_ratio/low_min": 7.509166607633233e-05, |
| "clip_ratio/region_mean": 0.0008942453190684318, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 920.0, |
| "completions/mean_length": 450.05999755859375, |
| "completions/mean_terminated_length": 348.0638122558594, |
| "completions/min_length": 151.0, |
| "completions/min_terminated_length": 151.0, |
| "entropy": 0.17726262360811235, |
| "epoch": 0.266304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017931127920746803, |
| "learning_rate": 5e-05, |
| "loss": 0.06858326494693756, |
| "num_tokens": 6114748.0, |
| "reward": 0.18024413287639618, |
| "reward_std": 0.6141991019248962, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.2197558581829071, |
| "rewards/length_penalty/std": 0.21670052409172058, |
| "sampling/importance_sampling_ratio/max": 2.825488328933716, |
| "sampling/importance_sampling_ratio/mean": 0.9923785924911499, |
| "sampling/importance_sampling_ratio/min": 0.27456802129745483, |
| "sampling/sampling_logp_difference/max": 1.2925561666488647, |
| "sampling/sampling_logp_difference/mean": 0.022753676399588585, |
| "step": 98, |
| "step_time": 21.57263088133186 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011265043169260025, |
| "clip_ratio/high_mean": 0.0011265043169260025, |
| "clip_ratio/low_mean": 0.00012555457651615144, |
| "clip_ratio/low_min": 0.00012555457651615144, |
| "clip_ratio/region_mean": 0.0012520588818006218, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 621.0, |
| "completions/max_terminated_length": 621.0, |
| "completions/mean_length": 328.4599914550781, |
| "completions/mean_terminated_length": 328.4599914550781, |
| "completions/min_length": 164.0, |
| "completions/min_terminated_length": 164.0, |
| "entropy": 0.1741026908159256, |
| "epoch": 0.26902173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012684342451393604, |
| "learning_rate": 5e-05, |
| "loss": 0.004149403423070908, |
| "num_tokens": 6134081.0, |
| "reward": 0.5996191501617432, |
| "reward_std": 0.4401145875453949, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.16038085520267487, |
| "rewards/length_penalty/std": 0.03834455832839012, |
| "sampling/importance_sampling_ratio/max": 2.659193754196167, |
| "sampling/importance_sampling_ratio/mean": 0.9934418201446533, |
| "sampling/importance_sampling_ratio/min": 0.3000729978084564, |
| "sampling/sampling_logp_difference/max": 1.203729510307312, |
| "sampling/sampling_logp_difference/mean": 0.026413219049572945, |
| "step": 99, |
| "step_time": 7.350566978333518 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010693204472772776, |
| "clip_ratio/high_mean": 0.0010693204472772776, |
| "clip_ratio/low_mean": 0.00018587360391393305, |
| "clip_ratio/low_min": 0.00018587360391393305, |
| "clip_ratio/region_mean": 0.0012551940511912108, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 691.0, |
| "completions/max_terminated_length": 691.0, |
| "completions/mean_length": 297.82000732421875, |
| "completions/mean_terminated_length": 297.82000732421875, |
| "completions/min_length": 161.0, |
| "completions/min_terminated_length": 161.0, |
| "entropy": 0.13965522944927217, |
| "epoch": 0.2717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0144702959805727, |
| "learning_rate": 5e-05, |
| "loss": 3.435742110013962e-05, |
| "num_tokens": 6151202.0, |
| "reward": 0.8145800828933716, |
| "reward_std": 0.20805624127388, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.1454199254512787, |
| "rewards/length_penalty/std": 0.046511922031641006, |
| "sampling/importance_sampling_ratio/max": 2.920515537261963, |
| "sampling/importance_sampling_ratio/mean": 0.9951541423797607, |
| "sampling/importance_sampling_ratio/min": 0.20719560980796814, |
| "sampling/sampling_logp_difference/max": 1.574091911315918, |
| "sampling/sampling_logp_difference/mean": 0.024871990084648132, |
| "step": 100, |
| "step_time": 7.712193023180589 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009782444452866913, |
| "clip_ratio/high_mean": 0.0009782444452866913, |
| "clip_ratio/low_mean": 5.1773234736174346e-05, |
| "clip_ratio/low_min": 5.1773234736174346e-05, |
| "clip_ratio/region_mean": 0.0010300176683813334, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1945.0, |
| "completions/mean_length": 465.3199768066406, |
| "completions/mean_terminated_length": 433.0203857421875, |
| "completions/min_length": 215.0, |
| "completions/min_terminated_length": 215.0, |
| "entropy": 0.18270143270492553, |
| "epoch": 0.27445652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.030248720198869705, |
| "learning_rate": 5e-05, |
| "loss": 0.08639093488454819, |
| "num_tokens": 6177358.0, |
| "reward": 0.7527929544448853, |
| "reward_std": 0.28967800736427307, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.2272070348262787, |
| "rewards/length_penalty/std": 0.1799187958240509, |
| "sampling/importance_sampling_ratio/max": 2.828981399536133, |
| "sampling/importance_sampling_ratio/mean": 0.9929043054580688, |
| "sampling/importance_sampling_ratio/min": 0.1369575411081314, |
| "sampling/sampling_logp_difference/max": 1.988084316253662, |
| "sampling/sampling_logp_difference/mean": 0.02395179495215416, |
| "step": 101, |
| "step_time": 22.081372380955145 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010954853089060635, |
| "clip_ratio/high_mean": 0.0010954853089060635, |
| "clip_ratio/low_mean": 0.00014803514641243966, |
| "clip_ratio/low_min": 0.00014803514641243966, |
| "clip_ratio/region_mean": 0.0012435204582288862, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1947.0, |
| "completions/max_terminated_length": 1947.0, |
| "completions/mean_length": 450.1399841308594, |
| "completions/mean_terminated_length": 450.1399841308594, |
| "completions/min_length": 131.0, |
| "completions/min_terminated_length": 131.0, |
| "entropy": 0.18313678503036498, |
| "epoch": 0.27717391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017789606004953384, |
| "learning_rate": 5e-05, |
| "loss": 0.01512880064547062, |
| "num_tokens": 6202795.0, |
| "reward": 0.5602050423622131, |
| "reward_std": 0.5041590929031372, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.21979492902755737, |
| "rewards/length_penalty/std": 0.16016612946987152, |
| "sampling/importance_sampling_ratio/max": 2.3117787837982178, |
| "sampling/importance_sampling_ratio/mean": 0.9927642345428467, |
| "sampling/importance_sampling_ratio/min": 0.20299819111824036, |
| "sampling/sampling_logp_difference/max": 1.5945582389831543, |
| "sampling/sampling_logp_difference/mean": 0.023115523159503937, |
| "step": 102, |
| "step_time": 20.302916617831215 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010320911649614572, |
| "clip_ratio/high_mean": 0.0010320911649614572, |
| "clip_ratio/low_mean": 0.0001328021287918091, |
| "clip_ratio/low_min": 0.0001328021287918091, |
| "clip_ratio/region_mean": 0.0011648932937532662, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 964.0, |
| "completions/max_terminated_length": 964.0, |
| "completions/mean_length": 339.2200012207031, |
| "completions/mean_terminated_length": 339.2200012207031, |
| "completions/min_length": 121.0, |
| "completions/min_terminated_length": 121.0, |
| "entropy": 0.22600021660327912, |
| "epoch": 0.2798913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017904600128531456, |
| "learning_rate": 5e-05, |
| "loss": 0.03389526531100273, |
| "num_tokens": 6223996.0, |
| "reward": 0.6343652009963989, |
| "reward_std": 0.4579472839832306, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.16563476622104645, |
| "rewards/length_penalty/std": 0.07587916404008865, |
| "sampling/importance_sampling_ratio/max": 2.5715408325195312, |
| "sampling/importance_sampling_ratio/mean": 0.9914016723632812, |
| "sampling/importance_sampling_ratio/min": 0.36589479446411133, |
| "sampling/sampling_logp_difference/max": 1.0054094791412354, |
| "sampling/sampling_logp_difference/mean": 0.030548855662345886, |
| "step": 103, |
| "step_time": 10.99179570376873 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009019060642458498, |
| "clip_ratio/high_mean": 0.0009019060642458498, |
| "clip_ratio/low_mean": 0.00018723605026025326, |
| "clip_ratio/low_min": 0.00018723605026025326, |
| "clip_ratio/region_mean": 0.00108914211159572, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1928.0, |
| "completions/mean_length": 758.719970703125, |
| "completions/mean_terminated_length": 705.0, |
| "completions/min_length": 216.0, |
| "completions/min_terminated_length": 216.0, |
| "entropy": 0.21980402171611785, |
| "epoch": 0.2826086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.029295017942786217, |
| "learning_rate": 5e-05, |
| "loss": 0.1281193643808365, |
| "num_tokens": 6265442.0, |
| "reward": -0.03046874888241291, |
| "reward_std": 0.5939843654632568, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.37046873569488525, |
| "rewards/length_penalty/std": 0.2411312311887741, |
| "sampling/importance_sampling_ratio/max": 2.8439342975616455, |
| "sampling/importance_sampling_ratio/mean": 0.9913606643676758, |
| "sampling/importance_sampling_ratio/min": 0.2924778461456299, |
| "sampling/sampling_logp_difference/max": 1.2293663024902344, |
| "sampling/sampling_logp_difference/mean": 0.02274427004158497, |
| "step": 104, |
| "step_time": 22.834440003614873 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011844986176583916, |
| "clip_ratio/high_mean": 0.0011844986176583916, |
| "clip_ratio/low_mean": 0.00021049597417004405, |
| "clip_ratio/low_min": 0.00021049597417004405, |
| "clip_ratio/region_mean": 0.0013949946092907338, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 611.0, |
| "completions/max_terminated_length": 611.0, |
| "completions/mean_length": 349.5, |
| "completions/mean_terminated_length": 349.5, |
| "completions/min_length": 123.0, |
| "completions/min_terminated_length": 123.0, |
| "entropy": 0.15829665660858155, |
| "epoch": 0.28532608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013679116033017635, |
| "learning_rate": 5e-05, |
| "loss": 0.003261825069785118, |
| "num_tokens": 6285277.0, |
| "reward": 0.4293456971645355, |
| "reward_std": 0.5380200147628784, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.170654296875, |
| "rewards/length_penalty/std": 0.0786299780011177, |
| "sampling/importance_sampling_ratio/max": 2.915198802947998, |
| "sampling/importance_sampling_ratio/mean": 0.9933813810348511, |
| "sampling/importance_sampling_ratio/min": 0.3001239597797394, |
| "sampling/sampling_logp_difference/max": 1.2035596370697021, |
| "sampling/sampling_logp_difference/mean": 0.021012984216213226, |
| "step": 105, |
| "step_time": 7.303402597783133 |
| }, |
| { |
| "clip_ratio/high_max": 0.001089644618332386, |
| "clip_ratio/high_mean": 0.001089644618332386, |
| "clip_ratio/low_mean": 0.00023785638040862978, |
| "clip_ratio/low_min": 0.00023785638040862978, |
| "clip_ratio/region_mean": 0.0013275009929202496, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 821.0, |
| "completions/max_terminated_length": 821.0, |
| "completions/mean_length": 323.8800048828125, |
| "completions/mean_terminated_length": 323.8800048828125, |
| "completions/min_length": 131.0, |
| "completions/min_terminated_length": 131.0, |
| "entropy": 0.13883249163627626, |
| "epoch": 0.28804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015942780300974846, |
| "learning_rate": 5e-05, |
| "loss": -0.004157735034823418, |
| "num_tokens": 6303951.0, |
| "reward": 0.76185542345047, |
| "reward_std": 0.31304851174354553, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.1581445336341858, |
| "rewards/length_penalty/std": 0.0903305858373642, |
| "sampling/importance_sampling_ratio/max": 2.1488256454467773, |
| "sampling/importance_sampling_ratio/mean": 0.9944878220558167, |
| "sampling/importance_sampling_ratio/min": 0.32026439905166626, |
| "sampling/sampling_logp_difference/max": 1.138608455657959, |
| "sampling/sampling_logp_difference/mean": 0.020255189388990402, |
| "step": 106, |
| "step_time": 9.11417300789617 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006249988509807736, |
| "clip_ratio/high_mean": 0.0006249988509807736, |
| "clip_ratio/low_mean": 0.00022756701218895614, |
| "clip_ratio/low_min": 0.00022756701218895614, |
| "clip_ratio/region_mean": 0.0008525658515281976, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1932.0, |
| "completions/mean_length": 664.0999755859375, |
| "completions/mean_terminated_length": 606.4375, |
| "completions/min_length": 174.0, |
| "completions/min_terminated_length": 174.0, |
| "entropy": 0.22402330338954926, |
| "epoch": 0.2907608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018222587183117867, |
| "learning_rate": 5e-05, |
| "loss": 0.03653242811560631, |
| "num_tokens": 6341196.0, |
| "reward": 0.2757324278354645, |
| "reward_std": 0.7114962935447693, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.32426756620407104, |
| "rewards/length_penalty/std": 0.27856025099754333, |
| "sampling/importance_sampling_ratio/max": 2.4848389625549316, |
| "sampling/importance_sampling_ratio/mean": 0.9914897680282593, |
| "sampling/importance_sampling_ratio/min": 0.1653730571269989, |
| "sampling/sampling_logp_difference/max": 1.7995513677597046, |
| "sampling/sampling_logp_difference/mean": 0.022073306143283844, |
| "step": 107, |
| "step_time": 23.499243413563818 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008343268476892262, |
| "clip_ratio/high_mean": 0.0008343268476892262, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008343268476892262, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 695.0, |
| "completions/max_terminated_length": 695.0, |
| "completions/mean_length": 361.47998046875, |
| "completions/mean_terminated_length": 361.47998046875, |
| "completions/min_length": 155.0, |
| "completions/min_terminated_length": 155.0, |
| "entropy": 0.20583085119724273, |
| "epoch": 0.29347826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01500002108514309, |
| "learning_rate": 5e-05, |
| "loss": -0.0029654093086719513, |
| "num_tokens": 6362320.0, |
| "reward": 0.3634960949420929, |
| "reward_std": 0.5250017642974854, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.17650391161441803, |
| "rewards/length_penalty/std": 0.07181988656520844, |
| "sampling/importance_sampling_ratio/max": 2.8058199882507324, |
| "sampling/importance_sampling_ratio/mean": 0.9923269748687744, |
| "sampling/importance_sampling_ratio/min": 0.19120995700359344, |
| "sampling/sampling_logp_difference/max": 1.6543831825256348, |
| "sampling/sampling_logp_difference/mean": 0.028827909380197525, |
| "step": 108, |
| "step_time": 8.322833819314837 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010915018559899182, |
| "clip_ratio/high_mean": 0.0010915018559899182, |
| "clip_ratio/low_mean": 0.00017255668935831637, |
| "clip_ratio/low_min": 0.00017255668935831637, |
| "clip_ratio/region_mean": 0.0012640585249755532, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1677.0, |
| "completions/mean_length": 442.7200012207031, |
| "completions/mean_terminated_length": 409.95916748046875, |
| "completions/min_length": 115.0, |
| "completions/min_terminated_length": 115.0, |
| "entropy": 0.26093956232070925, |
| "epoch": 0.296195652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01266445405781269, |
| "learning_rate": 5e-05, |
| "loss": 0.0036167525686323643, |
| "num_tokens": 6386796.0, |
| "reward": 0.46382811665534973, |
| "reward_std": 0.6575063467025757, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.47121208906173706, |
| "rewards/length_penalty/mean": -0.21617187559604645, |
| "rewards/length_penalty/std": 0.2383366823196411, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9895496368408203, |
| "sampling/importance_sampling_ratio/min": 0.29879647493362427, |
| "sampling/sampling_logp_difference/max": 1.207992672920227, |
| "sampling/sampling_logp_difference/mean": 0.028420479968190193, |
| "step": 109, |
| "step_time": 21.216636751545593 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011579408310353756, |
| "clip_ratio/high_mean": 0.0011579408310353756, |
| "clip_ratio/low_mean": 0.00013756073894910514, |
| "clip_ratio/low_min": 0.00013756073894910514, |
| "clip_ratio/region_mean": 0.0012955015816260129, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 502.0, |
| "completions/max_terminated_length": 502.0, |
| "completions/mean_length": 271.5799865722656, |
| "completions/mean_terminated_length": 271.5799865722656, |
| "completions/min_length": 106.0, |
| "completions/min_terminated_length": 106.0, |
| "entropy": 0.18488226532936097, |
| "epoch": 0.29891304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015168352983891964, |
| "learning_rate": 5e-05, |
| "loss": 0.012514734640717506, |
| "num_tokens": 6402955.0, |
| "reward": 0.8473925590515137, |
| "reward_std": 0.1584642231464386, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.13260741531848907, |
| "rewards/length_penalty/std": 0.05413118004798889, |
| "sampling/importance_sampling_ratio/max": 2.400756597518921, |
| "sampling/importance_sampling_ratio/mean": 0.992549479007721, |
| "sampling/importance_sampling_ratio/min": 0.2890666425228119, |
| "sampling/sampling_logp_difference/max": 1.2410980463027954, |
| "sampling/sampling_logp_difference/mean": 0.02747686579823494, |
| "step": 110, |
| "step_time": 6.049848190043122 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012651410943362862, |
| "clip_ratio/high_mean": 0.0012651410943362862, |
| "clip_ratio/low_mean": 8.190008229576051e-05, |
| "clip_ratio/low_min": 8.190008229576051e-05, |
| "clip_ratio/region_mean": 0.0013470411766320467, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 436.0, |
| "completions/max_terminated_length": 436.0, |
| "completions/mean_length": 247.1199951171875, |
| "completions/mean_terminated_length": 247.1199951171875, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.1295585960149765, |
| "epoch": 0.3016304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012275774963200092, |
| "learning_rate": 5e-05, |
| "loss": 0.008155894465744495, |
| "num_tokens": 6417971.0, |
| "reward": 0.8393359184265137, |
| "reward_std": 0.2072635143995285, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.12066406011581421, |
| "rewards/length_penalty/std": 0.04388388991355896, |
| "sampling/importance_sampling_ratio/max": 2.6247446537017822, |
| "sampling/importance_sampling_ratio/mean": 0.9953888058662415, |
| "sampling/importance_sampling_ratio/min": 0.1258077770471573, |
| "sampling/sampling_logp_difference/max": 2.073000192642212, |
| "sampling/sampling_logp_difference/mean": 0.022443437948822975, |
| "step": 111, |
| "step_time": 5.43325990694575 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008346656977664679, |
| "clip_ratio/high_mean": 0.0008346656977664679, |
| "clip_ratio/low_mean": 0.0005999092303682118, |
| "clip_ratio/low_min": 0.0005999092303682118, |
| "clip_ratio/region_mean": 0.0014345749048516154, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1531.0, |
| "completions/max_terminated_length": 1531.0, |
| "completions/mean_length": 310.1000061035156, |
| "completions/mean_terminated_length": 310.1000061035156, |
| "completions/min_length": 113.0, |
| "completions/min_terminated_length": 113.0, |
| "entropy": 0.22300356030464172, |
| "epoch": 0.30434782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01401439867913723, |
| "learning_rate": 5e-05, |
| "loss": 0.01661311835050583, |
| "num_tokens": 6435776.0, |
| "reward": 0.24858397245407104, |
| "reward_std": 0.5373894572257996, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.15141601860523224, |
| "rewards/length_penalty/std": 0.12549324333667755, |
| "sampling/importance_sampling_ratio/max": 2.561978578567505, |
| "sampling/importance_sampling_ratio/mean": 0.9911986589431763, |
| "sampling/importance_sampling_ratio/min": 0.23093871772289276, |
| "sampling/sampling_logp_difference/max": 1.4656028747558594, |
| "sampling/sampling_logp_difference/mean": 0.02530568838119507, |
| "step": 112, |
| "step_time": 16.101619800087065 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007494773017242551, |
| "clip_ratio/high_mean": 0.0007494773017242551, |
| "clip_ratio/low_mean": 0.0006496888410765678, |
| "clip_ratio/low_min": 0.0006496888410765678, |
| "clip_ratio/region_mean": 0.0013991661253385246, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 359.0, |
| "completions/max_terminated_length": 359.0, |
| "completions/mean_length": 185.25999450683594, |
| "completions/mean_terminated_length": 185.25999450683594, |
| "completions/min_length": 72.0, |
| "completions/min_terminated_length": 72.0, |
| "entropy": 0.1723326176404953, |
| "epoch": 0.3070652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02029413916170597, |
| "learning_rate": 5e-05, |
| "loss": 0.018048137426376343, |
| "num_tokens": 6448349.0, |
| "reward": 0.7095410227775574, |
| "reward_std": 0.42110520601272583, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.09045898169279099, |
| "rewards/length_penalty/std": 0.037427909672260284, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9949567317962646, |
| "sampling/importance_sampling_ratio/min": 0.21674247086048126, |
| "sampling/sampling_logp_difference/max": 1.5290454626083374, |
| "sampling/sampling_logp_difference/mean": 0.030676113441586494, |
| "step": 113, |
| "step_time": 4.628575901966542 |
| }, |
| { |
| "clip_ratio/high_max": 0.002369070309214294, |
| "clip_ratio/high_mean": 0.002369070309214294, |
| "clip_ratio/low_mean": 0.00010554089676588774, |
| "clip_ratio/low_min": 0.00010554089676588774, |
| "clip_ratio/region_mean": 0.0024746112292632462, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 350.0, |
| "completions/max_terminated_length": 350.0, |
| "completions/mean_length": 208.75999450683594, |
| "completions/mean_terminated_length": 208.75999450683594, |
| "completions/min_length": 80.0, |
| "completions/min_terminated_length": 80.0, |
| "entropy": 0.16646173298358918, |
| "epoch": 0.30978260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014696292579174042, |
| "learning_rate": 5e-05, |
| "loss": 0.002359196078032255, |
| "num_tokens": 6461327.0, |
| "reward": 0.7380663752555847, |
| "reward_std": 0.38239580392837524, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.10193359106779099, |
| "rewards/length_penalty/std": 0.03704632818698883, |
| "sampling/importance_sampling_ratio/max": 2.7964231967926025, |
| "sampling/importance_sampling_ratio/mean": 0.9943727254867554, |
| "sampling/importance_sampling_ratio/min": 0.24169418215751648, |
| "sampling/sampling_logp_difference/max": 1.4200820922851562, |
| "sampling/sampling_logp_difference/mean": 0.02667791210114956, |
| "step": 114, |
| "step_time": 4.516885473160073 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010958052007481456, |
| "clip_ratio/high_mean": 0.0010958052007481456, |
| "clip_ratio/low_mean": 0.0001959823537617922, |
| "clip_ratio/low_min": 0.0001959823537617922, |
| "clip_ratio/region_mean": 0.001291787507943809, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 515.0, |
| "completions/max_terminated_length": 515.0, |
| "completions/mean_length": 197.27999877929688, |
| "completions/mean_terminated_length": 197.27999877929688, |
| "completions/min_length": 88.0, |
| "completions/min_terminated_length": 88.0, |
| "entropy": 0.15682939291000367, |
| "epoch": 0.3125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013151104561984539, |
| "learning_rate": 5e-05, |
| "loss": -0.000307546928524971, |
| "num_tokens": 6473421.0, |
| "reward": 0.6836718320846558, |
| "reward_std": 0.4341813027858734, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.09632812440395355, |
| "rewards/length_penalty/std": 0.048083554953336716, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9945598840713501, |
| "sampling/importance_sampling_ratio/min": 0.19948126375675201, |
| "sampling/sampling_logp_difference/max": 1.6120349168777466, |
| "sampling/sampling_logp_difference/mean": 0.028728369623422623, |
| "step": 115, |
| "step_time": 5.90409521991387 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015390708926133812, |
| "clip_ratio/high_mean": 0.0015390708926133812, |
| "clip_ratio/low_mean": 0.0001479289960116148, |
| "clip_ratio/low_min": 0.0001479289960116148, |
| "clip_ratio/region_mean": 0.0016869998886249959, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 327.0, |
| "completions/max_terminated_length": 327.0, |
| "completions/mean_length": 170.3000030517578, |
| "completions/mean_terminated_length": 170.3000030517578, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.17938873767852784, |
| "epoch": 0.31521739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012773926369845867, |
| "learning_rate": 5e-05, |
| "loss": 0.0069333454594016075, |
| "num_tokens": 6484826.0, |
| "reward": 0.87684565782547, |
| "reward_std": 0.2142612338066101, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.08315429836511612, |
| "rewards/length_penalty/std": 0.03340579569339752, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9958755970001221, |
| "sampling/importance_sampling_ratio/min": 0.10664158314466476, |
| "sampling/sampling_logp_difference/max": 2.238281726837158, |
| "sampling/sampling_logp_difference/mean": 0.034581854939460754, |
| "step": 116, |
| "step_time": 4.133329353062436 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013375979149714112, |
| "clip_ratio/high_mean": 0.0013375979149714112, |
| "clip_ratio/low_mean": 0.0008156315423548221, |
| "clip_ratio/low_min": 0.0008156315423548221, |
| "clip_ratio/region_mean": 0.0021532294573262333, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 127.0, |
| "completions/max_terminated_length": 127.0, |
| "completions/mean_length": 74.4000015258789, |
| "completions/mean_terminated_length": 74.4000015258789, |
| "completions/min_length": 41.0, |
| "completions/min_terminated_length": 41.0, |
| "entropy": 0.19413722455501556, |
| "epoch": 0.3179347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01211152970790863, |
| "learning_rate": 5e-05, |
| "loss": 0.006493113934993744, |
| "num_tokens": 6490726.0, |
| "reward": 0.763671875, |
| "reward_std": 0.4076753854751587, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.03632812574505806, |
| "rewards/length_penalty/std": 0.009162030182778835, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927567839622498, |
| "sampling/importance_sampling_ratio/min": 0.023184923455119133, |
| "sampling/sampling_logp_difference/max": 3.7642531394958496, |
| "sampling/sampling_logp_difference/mean": 0.045659828931093216, |
| "step": 117, |
| "step_time": 2.2458587719593197 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011261911888141185, |
| "clip_ratio/high_mean": 0.0011261911888141185, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011261911888141185, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 292.0, |
| "completions/mean_length": 200.87998962402344, |
| "completions/mean_terminated_length": 163.1836700439453, |
| "completions/min_length": 61.0, |
| "completions/min_terminated_length": 61.0, |
| "entropy": 0.2687932997941971, |
| "epoch": 0.32065217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023787807673215866, |
| "learning_rate": 5e-05, |
| "loss": 0.06210978701710701, |
| "num_tokens": 6504450.0, |
| "reward": 0.7019140720367432, |
| "reward_std": 0.4698212444782257, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.09808593988418579, |
| "rewards/length_penalty/std": 0.13461492955684662, |
| "sampling/importance_sampling_ratio/max": 2.9681003093719482, |
| "sampling/importance_sampling_ratio/mean": 0.9874541163444519, |
| "sampling/importance_sampling_ratio/min": 0.1184043437242508, |
| "sampling/sampling_logp_difference/max": 2.1336498260498047, |
| "sampling/sampling_logp_difference/mean": 0.03763115033507347, |
| "step": 118, |
| "step_time": 20.160404477035627 |
| }, |
| { |
| "clip_ratio/high_max": 0.0017047577654011547, |
| "clip_ratio/high_mean": 0.0017047577654011547, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0017047577654011547, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 663.0, |
| "completions/max_terminated_length": 663.0, |
| "completions/mean_length": 197.6599884033203, |
| "completions/mean_terminated_length": 197.6599884033203, |
| "completions/min_length": 46.0, |
| "completions/min_terminated_length": 46.0, |
| "entropy": 0.18439604341983795, |
| "epoch": 0.3233695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010738166980445385, |
| "learning_rate": 5e-05, |
| "loss": -0.005624646320939064, |
| "num_tokens": 6517843.0, |
| "reward": 0.523486316204071, |
| "reward_std": 0.5189972519874573, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.09651367366313934, |
| "rewards/length_penalty/std": 0.06975338608026505, |
| "sampling/importance_sampling_ratio/max": 2.5998806953430176, |
| "sampling/importance_sampling_ratio/mean": 0.9924585819244385, |
| "sampling/importance_sampling_ratio/min": 0.2715196907520294, |
| "sampling/sampling_logp_difference/max": 1.3037207126617432, |
| "sampling/sampling_logp_difference/mean": 0.02511695772409439, |
| "step": 119, |
| "step_time": 7.475995167624205 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015201856731437147, |
| "clip_ratio/high_mean": 0.0015201856731437147, |
| "clip_ratio/low_mean": 0.0005623943405225873, |
| "clip_ratio/low_min": 0.0005623943405225873, |
| "clip_ratio/region_mean": 0.002082579943817109, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 454.0, |
| "completions/max_terminated_length": 454.0, |
| "completions/mean_length": 110.1199951171875, |
| "completions/mean_terminated_length": 110.1199951171875, |
| "completions/min_length": 44.0, |
| "completions/min_terminated_length": 44.0, |
| "entropy": 0.25055514872074125, |
| "epoch": 0.32608695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012562894262373447, |
| "learning_rate": 5e-05, |
| "loss": 0.0036469800397753716, |
| "num_tokens": 6526199.0, |
| "reward": 0.7862304449081421, |
| "reward_std": 0.37314942479133606, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.05376953259110451, |
| "rewards/length_penalty/std": 0.02968388795852661, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9884639382362366, |
| "sampling/importance_sampling_ratio/min": 0.10235190391540527, |
| "sampling/sampling_logp_difference/max": 2.2793383598327637, |
| "sampling/sampling_logp_difference/mean": 0.040974002331495285, |
| "step": 120, |
| "step_time": 4.958750201156363 |
| }, |
| { |
| "clip_ratio/high_max": 0.00166628155275248, |
| "clip_ratio/high_mean": 0.00166628155275248, |
| "clip_ratio/low_mean": 0.0004380452970508486, |
| "clip_ratio/low_min": 0.0004380452970508486, |
| "clip_ratio/region_mean": 0.0021043267683126033, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1279.0, |
| "completions/mean_length": 276.7200012207031, |
| "completions/mean_terminated_length": 163.65957641601562, |
| "completions/min_length": 22.0, |
| "completions/min_terminated_length": 22.0, |
| "entropy": 0.3713249295949936, |
| "epoch": 0.328804347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015158249996602535, |
| "learning_rate": 5e-05, |
| "loss": 0.01726599968969822, |
| "num_tokens": 6543375.0, |
| "reward": 0.4448828101158142, |
| "reward_std": 0.657846212387085, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.13511718809604645, |
| "rewards/length_penalty/std": 0.25230327248573303, |
| "sampling/importance_sampling_ratio/max": 2.5823233127593994, |
| "sampling/importance_sampling_ratio/mean": 0.9863095879554749, |
| "sampling/importance_sampling_ratio/min": 0.18198730051517487, |
| "sampling/sampling_logp_difference/max": 1.7038183212280273, |
| "sampling/sampling_logp_difference/mean": 0.03550250828266144, |
| "step": 121, |
| "step_time": 21.683151081902906 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015703110897447915, |
| "clip_ratio/high_mean": 0.0015703110897447915, |
| "clip_ratio/low_mean": 0.0003558584314305335, |
| "clip_ratio/low_min": 0.0003558584314305335, |
| "clip_ratio/region_mean": 0.0019261695502791553, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 969.0, |
| "completions/max_terminated_length": 969.0, |
| "completions/mean_length": 224.22000122070312, |
| "completions/mean_terminated_length": 224.22000122070312, |
| "completions/min_length": 34.0, |
| "completions/min_terminated_length": 34.0, |
| "entropy": 0.2967494040727615, |
| "epoch": 0.33152173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016116198152303696, |
| "learning_rate": 5e-05, |
| "loss": 0.0003641305956989527, |
| "num_tokens": 6557706.0, |
| "reward": 0.6505175828933716, |
| "reward_std": 0.4982417821884155, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.10948242247104645, |
| "rewards/length_penalty/std": 0.13545958697795868, |
| "sampling/importance_sampling_ratio/max": 2.4720304012298584, |
| "sampling/importance_sampling_ratio/mean": 0.9879042506217957, |
| "sampling/importance_sampling_ratio/min": 0.2803405225276947, |
| "sampling/sampling_logp_difference/max": 1.2717502117156982, |
| "sampling/sampling_logp_difference/mean": 0.033322110772132874, |
| "step": 122, |
| "step_time": 10.36594293313101 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015112032764591278, |
| "clip_ratio/high_mean": 0.0015112032764591278, |
| "clip_ratio/low_mean": 4.3075598659925164e-05, |
| "clip_ratio/low_min": 4.3075598659925164e-05, |
| "clip_ratio/region_mean": 0.001554278878029436, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1715.0, |
| "completions/mean_length": 194.63999938964844, |
| "completions/mean_terminated_length": 156.8163299560547, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.23585395216941835, |
| "epoch": 0.3342391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022392380982637405, |
| "learning_rate": 5e-05, |
| "loss": 0.036236222833395004, |
| "num_tokens": 6570678.0, |
| "reward": 0.5449609160423279, |
| "reward_std": 0.5550513863563538, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.09503906220197678, |
| "rewards/length_penalty/std": 0.1725759208202362, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9900131225585938, |
| "sampling/importance_sampling_ratio/min": 0.20634840428829193, |
| "sampling/sampling_logp_difference/max": 1.7392792701721191, |
| "sampling/sampling_logp_difference/mean": 0.030603135004639626, |
| "step": 123, |
| "step_time": 20.26159128616564 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005344107747077941, |
| "clip_ratio/high_mean": 0.0005344107747077941, |
| "clip_ratio/low_mean": 0.00024125452619045974, |
| "clip_ratio/low_min": 0.00024125452619045974, |
| "clip_ratio/region_mean": 0.0007756653008982539, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 194.0, |
| "completions/max_terminated_length": 194.0, |
| "completions/mean_length": 82.18000030517578, |
| "completions/mean_terminated_length": 82.18000030517578, |
| "completions/min_length": 25.0, |
| "completions/min_terminated_length": 25.0, |
| "entropy": 0.1568037450313568, |
| "epoch": 0.33695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014341763220727444, |
| "learning_rate": 5e-05, |
| "loss": 0.0004445682279765606, |
| "num_tokens": 6577037.0, |
| "reward": 0.6598730087280273, |
| "reward_std": 0.4739854633808136, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.04012695327401161, |
| "rewards/length_penalty/std": 0.020628858357667923, |
| "sampling/importance_sampling_ratio/max": 2.8568575382232666, |
| "sampling/importance_sampling_ratio/mean": 0.9948601126670837, |
| "sampling/importance_sampling_ratio/min": 0.4282723367214203, |
| "sampling/sampling_logp_difference/max": 1.0497221946716309, |
| "sampling/sampling_logp_difference/mean": 0.02735116146504879, |
| "step": 124, |
| "step_time": 2.682286828290671 |
| }, |
| { |
| "clip_ratio/high_max": 0.0021309208823367953, |
| "clip_ratio/high_mean": 0.0021309208823367953, |
| "clip_ratio/low_mean": 0.0005215916316956282, |
| "clip_ratio/low_min": 0.0005215916316956282, |
| "clip_ratio/region_mean": 0.0026525125140324235, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 176.0, |
| "completions/max_terminated_length": 176.0, |
| "completions/mean_length": 80.29999542236328, |
| "completions/mean_terminated_length": 80.29999542236328, |
| "completions/min_length": 30.0, |
| "completions/min_terminated_length": 30.0, |
| "entropy": 0.21028359234333038, |
| "epoch": 0.33967391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011741945520043373, |
| "learning_rate": 5e-05, |
| "loss": -0.0010668456088751554, |
| "num_tokens": 6583532.0, |
| "reward": 0.5407909750938416, |
| "reward_std": 0.5034129619598389, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.03920898586511612, |
| "rewards/length_penalty/std": 0.019870659336447716, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9912551045417786, |
| "sampling/importance_sampling_ratio/min": 0.24525412917137146, |
| "sampling/sampling_logp_difference/max": 1.4401545524597168, |
| "sampling/sampling_logp_difference/mean": 0.03436288982629776, |
| "step": 125, |
| "step_time": 2.536663970211521 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011726996744982898, |
| "clip_ratio/high_mean": 0.0011726996744982898, |
| "clip_ratio/low_mean": 0.00038506463170051575, |
| "clip_ratio/low_min": 0.00038506463170051575, |
| "clip_ratio/region_mean": 0.0015577643178403377, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2036.0, |
| "completions/max_terminated_length": 2036.0, |
| "completions/mean_length": 239.5, |
| "completions/mean_terminated_length": 239.5, |
| "completions/min_length": 38.0, |
| "completions/min_terminated_length": 38.0, |
| "entropy": 0.30440682768821714, |
| "epoch": 0.3423913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018481383100152016, |
| "learning_rate": 5e-05, |
| "loss": -0.016416456550359726, |
| "num_tokens": 6600477.0, |
| "reward": 0.383056640625, |
| "reward_std": 0.5394988656044006, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.116943359375, |
| "rewards/length_penalty/std": 0.15436860918998718, |
| "sampling/importance_sampling_ratio/max": 2.092435121536255, |
| "sampling/importance_sampling_ratio/mean": 0.9891465306282043, |
| "sampling/importance_sampling_ratio/min": 0.353488564491272, |
| "sampling/sampling_logp_difference/max": 1.0399041175842285, |
| "sampling/sampling_logp_difference/mean": 0.030273286625742912, |
| "step": 126, |
| "step_time": 20.4199744600337 |
| }, |
| { |
| "clip_ratio/high_max": 0.0021248800330795348, |
| "clip_ratio/high_mean": 0.0021248800330795348, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0021248800330795348, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 328.0, |
| "completions/max_terminated_length": 328.0, |
| "completions/mean_length": 116.5999984741211, |
| "completions/mean_terminated_length": 116.5999984741211, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.21633773446083068, |
| "epoch": 0.3451086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01247034128755331, |
| "learning_rate": 5e-05, |
| "loss": -0.0008256335277110338, |
| "num_tokens": 6608937.0, |
| "reward": 0.7630664110183716, |
| "reward_std": 0.3820847272872925, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.05693359300494194, |
| "rewards/length_penalty/std": 0.03554932400584221, |
| "sampling/importance_sampling_ratio/max": 2.8851640224456787, |
| "sampling/importance_sampling_ratio/mean": 0.9902939796447754, |
| "sampling/importance_sampling_ratio/min": 0.14913764595985413, |
| "sampling/sampling_logp_difference/max": 1.9028856754302979, |
| "sampling/sampling_logp_difference/mean": 0.03269968554377556, |
| "step": 127, |
| "step_time": 3.947012424468994 |
| }, |
| { |
| "clip_ratio/high_max": 0.00036160510499030354, |
| "clip_ratio/high_mean": 0.00036160510499030354, |
| "clip_ratio/low_mean": 0.0002602472435683012, |
| "clip_ratio/low_min": 0.0002602472435683012, |
| "clip_ratio/region_mean": 0.0006218523252755403, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 532.0, |
| "completions/max_terminated_length": 532.0, |
| "completions/mean_length": 94.18000030517578, |
| "completions/mean_terminated_length": 94.18000030517578, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.23007185161113738, |
| "epoch": 0.34782608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017713308334350586, |
| "learning_rate": 5e-05, |
| "loss": 0.015209197998046875, |
| "num_tokens": 6615966.0, |
| "reward": 0.8940136432647705, |
| "reward_std": 0.26931431889533997, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.04598632827401161, |
| "rewards/length_penalty/std": 0.04119149595499039, |
| "sampling/importance_sampling_ratio/max": 2.860724925994873, |
| "sampling/importance_sampling_ratio/mean": 0.9925522804260254, |
| "sampling/importance_sampling_ratio/min": 0.3146184980869293, |
| "sampling/sampling_logp_difference/max": 1.1563944816589355, |
| "sampling/sampling_logp_difference/mean": 0.03633886203169823, |
| "step": 128, |
| "step_time": 5.5231674641836435 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007146310119424016, |
| "clip_ratio/high_mean": 0.0007146310119424016, |
| "clip_ratio/low_mean": 0.0010484752245247364, |
| "clip_ratio/low_min": 0.0010484752245247364, |
| "clip_ratio/region_mean": 0.0017631062073633075, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 476.0, |
| "completions/mean_length": 172.4199981689453, |
| "completions/mean_terminated_length": 134.14285278320312, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.2042074203491211, |
| "epoch": 0.35054347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016879143193364143, |
| "learning_rate": 5e-05, |
| "loss": 0.03205213323235512, |
| "num_tokens": 6627647.0, |
| "reward": 0.3558105528354645, |
| "reward_std": 0.5454025864601135, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.08418945223093033, |
| "rewards/length_penalty/std": 0.1382221281528473, |
| "sampling/importance_sampling_ratio/max": 2.593198537826538, |
| "sampling/importance_sampling_ratio/mean": 0.9910122752189636, |
| "sampling/importance_sampling_ratio/min": 0.2354326844215393, |
| "sampling/sampling_logp_difference/max": 1.4463303089141846, |
| "sampling/sampling_logp_difference/mean": 0.02556837536394596, |
| "step": 129, |
| "step_time": 20.19545165193267 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007989593432284892, |
| "clip_ratio/high_mean": 0.0007989593432284892, |
| "clip_ratio/low_mean": 0.0006685091881081462, |
| "clip_ratio/low_min": 0.0006685091881081462, |
| "clip_ratio/region_mean": 0.001467468508053571, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 777.0, |
| "completions/max_terminated_length": 777.0, |
| "completions/mean_length": 142.5399932861328, |
| "completions/mean_terminated_length": 142.5399932861328, |
| "completions/min_length": 39.0, |
| "completions/min_terminated_length": 39.0, |
| "entropy": 0.382894241809845, |
| "epoch": 0.3532608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016726333647966385, |
| "learning_rate": 5e-05, |
| "loss": 0.006665468215942383, |
| "num_tokens": 6637544.0, |
| "reward": 0.7504003643989563, |
| "reward_std": 0.4603516757488251, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.06959960609674454, |
| "rewards/length_penalty/std": 0.08425810188055038, |
| "sampling/importance_sampling_ratio/max": 1.9820367097854614, |
| "sampling/importance_sampling_ratio/mean": 0.9842870235443115, |
| "sampling/importance_sampling_ratio/min": 0.25300493836402893, |
| "sampling/sampling_logp_difference/max": 1.3743462562561035, |
| "sampling/sampling_logp_difference/mean": 0.03975590690970421, |
| "step": 130, |
| "step_time": 8.106362144928426 |
| }, |
| { |
| "clip_ratio/high_max": 0.001576645253226161, |
| "clip_ratio/high_mean": 0.001576645253226161, |
| "clip_ratio/low_mean": 0.00030627872329205277, |
| "clip_ratio/low_min": 0.00030627872329205277, |
| "clip_ratio/region_mean": 0.0018829239532351493, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 376.0, |
| "completions/max_terminated_length": 376.0, |
| "completions/mean_length": 134.72000122070312, |
| "completions/mean_terminated_length": 134.72000122070312, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.28692707419395447, |
| "epoch": 0.35597826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017507078126072884, |
| "learning_rate": 5e-05, |
| "loss": 0.005320252384990454, |
| "num_tokens": 6647660.0, |
| "reward": 0.6342187523841858, |
| "reward_std": 0.47180676460266113, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.06578125059604645, |
| "rewards/length_penalty/std": 0.048506397753953934, |
| "sampling/importance_sampling_ratio/max": 2.642646312713623, |
| "sampling/importance_sampling_ratio/mean": 0.9907641410827637, |
| "sampling/importance_sampling_ratio/min": 0.2680867612361908, |
| "sampling/sampling_logp_difference/max": 1.3164446353912354, |
| "sampling/sampling_logp_difference/mean": 0.03497415781021118, |
| "step": 131, |
| "step_time": 4.654536312445998 |
| }, |
| { |
| "clip_ratio/high_max": 0.0025257899891585112, |
| "clip_ratio/high_mean": 0.0025257899891585112, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0025257899891585112, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 163.0, |
| "completions/mean_length": 112.73999786376953, |
| "completions/mean_terminated_length": 73.2448959350586, |
| "completions/min_length": 31.0, |
| "completions/min_terminated_length": 31.0, |
| "entropy": 0.15580243691802026, |
| "epoch": 0.358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008819308131933212, |
| "learning_rate": 5e-05, |
| "loss": -0.002805797616019845, |
| "num_tokens": 6655127.0, |
| "reward": 0.6849511861801147, |
| "reward_std": 0.45300376415252686, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.055048827081918716, |
| "rewards/length_penalty/std": 0.13734318315982819, |
| "sampling/importance_sampling_ratio/max": 2.317105293273926, |
| "sampling/importance_sampling_ratio/mean": 0.9940555095672607, |
| "sampling/importance_sampling_ratio/min": 0.3421003520488739, |
| "sampling/sampling_logp_difference/max": 1.0726511478424072, |
| "sampling/sampling_logp_difference/mean": 0.018943725153803825, |
| "step": 132, |
| "step_time": 19.143756069708616 |
| }, |
| { |
| "clip_ratio/high_max": 0.002107415325008333, |
| "clip_ratio/high_mean": 0.002107415325008333, |
| "clip_ratio/low_mean": 0.0004080776940099895, |
| "clip_ratio/low_min": 0.0004080776940099895, |
| "clip_ratio/region_mean": 0.002515493053942919, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 239.0, |
| "completions/max_terminated_length": 239.0, |
| "completions/mean_length": 99.22000122070312, |
| "completions/mean_terminated_length": 99.22000122070312, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.2283396154642105, |
| "epoch": 0.36141304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010727017186582088, |
| "learning_rate": 5e-05, |
| "loss": 0.0031031088437885046, |
| "num_tokens": 6662778.0, |
| "reward": 0.9115527272224426, |
| "reward_std": 0.20123639702796936, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.04844726622104645, |
| "rewards/length_penalty/std": 0.026893993839621544, |
| "sampling/importance_sampling_ratio/max": 2.5560829639434814, |
| "sampling/importance_sampling_ratio/mean": 0.9919185638427734, |
| "sampling/importance_sampling_ratio/min": 0.3740426003932953, |
| "sampling/sampling_logp_difference/max": 0.9833855628967285, |
| "sampling/sampling_logp_difference/mean": 0.031429849565029144, |
| "step": 133, |
| "step_time": 3.1372911932412535 |
| }, |
| { |
| "clip_ratio/high_max": 0.001948577049188316, |
| "clip_ratio/high_mean": 0.001948577049188316, |
| "clip_ratio/low_mean": 0.00043497568112798034, |
| "clip_ratio/low_min": 0.00043497568112798034, |
| "clip_ratio/region_mean": 0.0023835527477785944, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 904.0, |
| "completions/mean_length": 170.63999938964844, |
| "completions/mean_terminated_length": 132.32652282714844, |
| "completions/min_length": 38.0, |
| "completions/min_terminated_length": 38.0, |
| "entropy": 0.2732980579137802, |
| "epoch": 0.3641304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018917858600616455, |
| "learning_rate": 5e-05, |
| "loss": 0.027346350252628326, |
| "num_tokens": 6674000.0, |
| "reward": 0.8766796588897705, |
| "reward_std": 0.27942439913749695, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.08332031220197678, |
| "rewards/length_penalty/std": 0.15983402729034424, |
| "sampling/importance_sampling_ratio/max": 2.3661015033721924, |
| "sampling/importance_sampling_ratio/mean": 0.9905344843864441, |
| "sampling/importance_sampling_ratio/min": 0.35583624243736267, |
| "sampling/sampling_logp_difference/max": 1.0332846641540527, |
| "sampling/sampling_logp_difference/mean": 0.024907777085900307, |
| "step": 134, |
| "step_time": 19.943776425207034 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011135154054500163, |
| "clip_ratio/high_mean": 0.0011135154054500163, |
| "clip_ratio/low_mean": 0.00042200194438919426, |
| "clip_ratio/low_min": 0.00042200194438919426, |
| "clip_ratio/region_mean": 0.0015355173381976783, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1089.0, |
| "completions/mean_length": 316.6399841308594, |
| "completions/mean_terminated_length": 206.12765502929688, |
| "completions/min_length": 61.0, |
| "completions/min_terminated_length": 61.0, |
| "entropy": 0.29149270951747897, |
| "epoch": 0.36684782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01800808683037758, |
| "learning_rate": 5e-05, |
| "loss": 0.03891075402498245, |
| "num_tokens": 6694012.0, |
| "reward": 0.3053906261920929, |
| "reward_std": 0.6115123629570007, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.15460938215255737, |
| "rewards/length_penalty/std": 0.2331872135400772, |
| "sampling/importance_sampling_ratio/max": 2.527808904647827, |
| "sampling/importance_sampling_ratio/mean": 0.9900528192520142, |
| "sampling/importance_sampling_ratio/min": 0.3485993444919586, |
| "sampling/sampling_logp_difference/max": 1.053831934928894, |
| "sampling/sampling_logp_difference/mean": 0.027457833290100098, |
| "step": 135, |
| "step_time": 21.226410260191187 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006706602172926069, |
| "clip_ratio/high_mean": 0.0006706602172926069, |
| "clip_ratio/low_mean": 0.0006880840577650815, |
| "clip_ratio/low_min": 0.0006880840577650815, |
| "clip_ratio/region_mean": 0.0013587442692369223, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 213.0, |
| "completions/mean_length": 184.17999267578125, |
| "completions/mean_terminated_length": 106.52083587646484, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.2364450290799141, |
| "epoch": 0.3695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017749518156051636, |
| "learning_rate": 5e-05, |
| "loss": 0.04006284475326538, |
| "num_tokens": 6705301.0, |
| "reward": 0.5100683569908142, |
| "reward_std": 0.5728880167007446, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.08993163704872131, |
| "rewards/length_penalty/std": 0.18873944878578186, |
| "sampling/importance_sampling_ratio/max": 2.6783275604248047, |
| "sampling/importance_sampling_ratio/mean": 0.9919785857200623, |
| "sampling/importance_sampling_ratio/min": 0.25271889567375183, |
| "sampling/sampling_logp_difference/max": 1.3754775524139404, |
| "sampling/sampling_logp_difference/mean": 0.024434348568320274, |
| "step": 136, |
| "step_time": 20.306161327054724 |
| }, |
| { |
| "clip_ratio/high_max": 0.001449799770489335, |
| "clip_ratio/high_mean": 0.001449799770489335, |
| "clip_ratio/low_mean": 0.0008101428858935833, |
| "clip_ratio/low_min": 0.0008101428858935833, |
| "clip_ratio/region_mean": 0.0022599426563829185, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 294.0, |
| "completions/max_terminated_length": 294.0, |
| "completions/mean_length": 107.13999938964844, |
| "completions/mean_terminated_length": 107.13999938964844, |
| "completions/min_length": 25.0, |
| "completions/min_terminated_length": 25.0, |
| "entropy": 0.19550255239009856, |
| "epoch": 0.37228260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007440829183906317, |
| "learning_rate": 5e-05, |
| "loss": 0.0010275387903675437, |
| "num_tokens": 6713518.0, |
| "reward": 0.32768553495407104, |
| "reward_std": 0.5096645951271057, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.052314452826976776, |
| "rewards/length_penalty/std": 0.02687421627342701, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9913536906242371, |
| "sampling/importance_sampling_ratio/min": 0.3674847185611725, |
| "sampling/sampling_logp_difference/max": 1.5071141719818115, |
| "sampling/sampling_logp_difference/mean": 0.03210548311471939, |
| "step": 137, |
| "step_time": 4.197776689194143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013970615924336015, |
| "clip_ratio/high_mean": 0.0013970615924336015, |
| "clip_ratio/low_mean": 0.0004110713372938335, |
| "clip_ratio/low_min": 0.0004110713372938335, |
| "clip_ratio/region_mean": 0.0018081329413689672, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 618.0, |
| "completions/max_terminated_length": 618.0, |
| "completions/mean_length": 146.05999755859375, |
| "completions/mean_terminated_length": 146.05999755859375, |
| "completions/min_length": 40.0, |
| "completions/min_terminated_length": 40.0, |
| "entropy": 0.23566215634346008, |
| "epoch": 0.375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013943018391728401, |
| "learning_rate": 5e-05, |
| "loss": 0.018286939710378647, |
| "num_tokens": 6723851.0, |
| "reward": 0.5286816358566284, |
| "reward_std": 0.5490127801895142, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.0713183581829071, |
| "rewards/length_penalty/std": 0.07072754204273224, |
| "sampling/importance_sampling_ratio/max": 2.412440299987793, |
| "sampling/importance_sampling_ratio/mean": 0.9919129610061646, |
| "sampling/importance_sampling_ratio/min": 0.24572424590587616, |
| "sampling/sampling_logp_difference/max": 1.4035453796386719, |
| "sampling/sampling_logp_difference/mean": 0.03258337453007698, |
| "step": 138, |
| "step_time": 6.693183201830834 |
| }, |
| { |
| "clip_ratio/high_max": 0.002049162099137902, |
| "clip_ratio/high_mean": 0.002049162099137902, |
| "clip_ratio/low_mean": 0.0003365680342540145, |
| "clip_ratio/low_min": 0.0003365680342540145, |
| "clip_ratio/region_mean": 0.002385730156674981, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 850.0, |
| "completions/max_terminated_length": 850.0, |
| "completions/mean_length": 152.05999755859375, |
| "completions/mean_terminated_length": 152.05999755859375, |
| "completions/min_length": 28.0, |
| "completions/min_terminated_length": 28.0, |
| "entropy": 0.3924503058195114, |
| "epoch": 0.37771739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020034583285450935, |
| "learning_rate": 5e-05, |
| "loss": -0.017497628927230835, |
| "num_tokens": 6734234.0, |
| "reward": 0.5057519674301147, |
| "reward_std": 0.5376050472259521, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.0742480456829071, |
| "rewards/length_penalty/std": 0.09696489572525024, |
| "sampling/importance_sampling_ratio/max": 2.3507442474365234, |
| "sampling/importance_sampling_ratio/mean": 0.983169436454773, |
| "sampling/importance_sampling_ratio/min": 0.3438117504119873, |
| "sampling/sampling_logp_difference/max": 1.0676610469818115, |
| "sampling/sampling_logp_difference/mean": 0.0407370924949646, |
| "step": 139, |
| "step_time": 8.90356107102707 |
| }, |
| { |
| "clip_ratio/high_max": 0.0033710386953316627, |
| "clip_ratio/high_mean": 0.0033710386953316627, |
| "clip_ratio/low_mean": 0.0003203466010745615, |
| "clip_ratio/low_min": 0.0003203466010745615, |
| "clip_ratio/region_mean": 0.0036913853138685225, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1558.0, |
| "completions/max_terminated_length": 1558.0, |
| "completions/mean_length": 167.6199951171875, |
| "completions/mean_terminated_length": 167.6199951171875, |
| "completions/min_length": 38.0, |
| "completions/min_terminated_length": 38.0, |
| "entropy": 0.5172111749649048, |
| "epoch": 0.3804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015268613584339619, |
| "learning_rate": 5e-05, |
| "loss": 0.0037686056457459927, |
| "num_tokens": 6746255.0, |
| "reward": 0.6381542682647705, |
| "reward_std": 0.5382852554321289, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573422908783, |
| "rewards/length_penalty/mean": -0.08184570074081421, |
| "rewards/length_penalty/std": 0.13116991519927979, |
| "sampling/importance_sampling_ratio/max": 2.9720616340637207, |
| "sampling/importance_sampling_ratio/mean": 0.9806302785873413, |
| "sampling/importance_sampling_ratio/min": 0.23781552910804749, |
| "sampling/sampling_logp_difference/max": 1.4362599849700928, |
| "sampling/sampling_logp_difference/mean": 0.04316100478172302, |
| "step": 140, |
| "step_time": 15.562143780989572 |
| }, |
| { |
| "clip_ratio/high_max": 0.001871592167299241, |
| "clip_ratio/high_mean": 0.001871592167299241, |
| "clip_ratio/low_mean": 0.0003299024887382984, |
| "clip_ratio/low_min": 0.0003299024887382984, |
| "clip_ratio/region_mean": 0.002201494702603668, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 426.0, |
| "completions/max_terminated_length": 426.0, |
| "completions/mean_length": 108.79999542236328, |
| "completions/mean_terminated_length": 108.79999542236328, |
| "completions/min_length": 21.0, |
| "completions/min_terminated_length": 21.0, |
| "entropy": 0.22569895684719085, |
| "epoch": 0.38315217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007933283224701881, |
| "learning_rate": 5e-05, |
| "loss": -6.703822873532772e-05, |
| "num_tokens": 6754425.0, |
| "reward": 0.7068749666213989, |
| "reward_std": 0.42361482977867126, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.05312500149011612, |
| "rewards/length_penalty/std": 0.04547841474413872, |
| "sampling/importance_sampling_ratio/max": 2.0014195442199707, |
| "sampling/importance_sampling_ratio/mean": 0.9917043447494507, |
| "sampling/importance_sampling_ratio/min": 0.3356453776359558, |
| "sampling/sampling_logp_difference/max": 1.0917000770568848, |
| "sampling/sampling_logp_difference/mean": 0.029148748144507408, |
| "step": 141, |
| "step_time": 4.8307973828632385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014908161596395076, |
| "clip_ratio/high_mean": 0.0014908161596395076, |
| "clip_ratio/low_mean": 0.00033613445702940226, |
| "clip_ratio/low_min": 0.00033613445702940226, |
| "clip_ratio/region_mean": 0.0018269506166689099, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 474.0, |
| "completions/max_terminated_length": 474.0, |
| "completions/mean_length": 134.17999267578125, |
| "completions/mean_terminated_length": 134.17999267578125, |
| "completions/min_length": 22.0, |
| "completions/min_terminated_length": 22.0, |
| "entropy": 0.24973546266555785, |
| "epoch": 0.3858695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009919676929712296, |
| "learning_rate": 5e-05, |
| "loss": 0.0027852458879351616, |
| "num_tokens": 6763924.0, |
| "reward": 0.7744824290275574, |
| "reward_std": 0.368818998336792, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.06551757454872131, |
| "rewards/length_penalty/std": 0.046280987560749054, |
| "sampling/importance_sampling_ratio/max": 2.9000210762023926, |
| "sampling/importance_sampling_ratio/mean": 0.9904726147651672, |
| "sampling/importance_sampling_ratio/min": 0.33445116877555847, |
| "sampling/sampling_logp_difference/max": 1.0952644348144531, |
| "sampling/sampling_logp_difference/mean": 0.03229127079248428, |
| "step": 142, |
| "step_time": 5.309932762756944 |
| }, |
| { |
| "clip_ratio/high_max": 0.0023459566989913585, |
| "clip_ratio/high_mean": 0.0023459566989913585, |
| "clip_ratio/low_mean": 0.0002624671906232834, |
| "clip_ratio/low_min": 0.0002624671906232834, |
| "clip_ratio/region_mean": 0.002608423889614642, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 191.0, |
| "completions/max_terminated_length": 191.0, |
| "completions/mean_length": 88.31999969482422, |
| "completions/mean_terminated_length": 88.31999969482422, |
| "completions/min_length": 16.0, |
| "completions/min_terminated_length": 16.0, |
| "entropy": 0.1915595978498459, |
| "epoch": 0.38858695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009547167457640171, |
| "learning_rate": 5e-05, |
| "loss": 0.002988064894452691, |
| "num_tokens": 6770680.0, |
| "reward": 0.6768749952316284, |
| "reward_std": 0.4728812575340271, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.04312499985098839, |
| "rewards/length_penalty/std": 0.028845716267824173, |
| "sampling/importance_sampling_ratio/max": 1.9959818124771118, |
| "sampling/importance_sampling_ratio/mean": 0.9905638694763184, |
| "sampling/importance_sampling_ratio/min": 0.38347241282463074, |
| "sampling/sampling_logp_difference/max": 0.9584876298904419, |
| "sampling/sampling_logp_difference/mean": 0.025498980656266212, |
| "step": 143, |
| "step_time": 2.71979623218067 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007215470774099231, |
| "clip_ratio/high_mean": 0.0007215470774099231, |
| "clip_ratio/low_mean": 0.000610530423000455, |
| "clip_ratio/low_min": 0.000610530423000455, |
| "clip_ratio/region_mean": 0.0013320774887688458, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1141.0, |
| "completions/max_terminated_length": 1141.0, |
| "completions/mean_length": 200.37998962402344, |
| "completions/mean_terminated_length": 200.37998962402344, |
| "completions/min_length": 22.0, |
| "completions/min_terminated_length": 22.0, |
| "entropy": 0.3568861961364746, |
| "epoch": 0.391304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022187354043126106, |
| "learning_rate": 5e-05, |
| "loss": 0.0005749613046646118, |
| "num_tokens": 6783749.0, |
| "reward": 0.462158203125, |
| "reward_std": 0.551166832447052, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.09784179925918579, |
| "rewards/length_penalty/std": 0.1214737519621849, |
| "sampling/importance_sampling_ratio/max": 2.6011152267456055, |
| "sampling/importance_sampling_ratio/mean": 0.9874166250228882, |
| "sampling/importance_sampling_ratio/min": 0.45548123121261597, |
| "sampling/sampling_logp_difference/max": 0.9559402465820312, |
| "sampling/sampling_logp_difference/mean": 0.03136809170246124, |
| "step": 144, |
| "step_time": 11.803367036161944 |
| }, |
| { |
| "clip_ratio/high_max": 0.0017857325496152044, |
| "clip_ratio/high_mean": 0.0017857325496152044, |
| "clip_ratio/low_mean": 0.001115760114043951, |
| "clip_ratio/low_min": 0.001115760114043951, |
| "clip_ratio/region_mean": 0.0029014926869422196, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 202.0, |
| "completions/max_terminated_length": 202.0, |
| "completions/mean_length": 73.0999984741211, |
| "completions/mean_terminated_length": 73.0999984741211, |
| "completions/min_length": 21.0, |
| "completions/min_terminated_length": 21.0, |
| "entropy": 0.26529322266578675, |
| "epoch": 0.39402173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012280095368623734, |
| "learning_rate": 5e-05, |
| "loss": 0.01031698938459158, |
| "num_tokens": 6789734.0, |
| "reward": 0.5643066167831421, |
| "reward_std": 0.4961875379085541, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.03569335862994194, |
| "rewards/length_penalty/std": 0.023383095860481262, |
| "sampling/importance_sampling_ratio/max": 2.72530460357666, |
| "sampling/importance_sampling_ratio/mean": 0.9867799878120422, |
| "sampling/importance_sampling_ratio/min": 0.3227711021900177, |
| "sampling/sampling_logp_difference/max": 1.1308119297027588, |
| "sampling/sampling_logp_difference/mean": 0.0355399064719677, |
| "step": 145, |
| "step_time": 2.714483188930899 |
| }, |
| { |
| "clip_ratio/high_max": 0.003725502814631909, |
| "clip_ratio/high_mean": 0.003725502814631909, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.003725502814631909, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 260.0, |
| "completions/max_terminated_length": 260.0, |
| "completions/mean_length": 104.0999984741211, |
| "completions/mean_terminated_length": 104.0999984741211, |
| "completions/min_length": 24.0, |
| "completions/min_terminated_length": 24.0, |
| "entropy": 0.2583618342876434, |
| "epoch": 0.3967391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014577813446521759, |
| "learning_rate": 5e-05, |
| "loss": -0.004070304799824953, |
| "num_tokens": 6797879.0, |
| "reward": 0.7891699075698853, |
| "reward_std": 0.37547630071640015, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.05083007737994194, |
| "rewards/length_penalty/std": 0.03442299738526344, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9909201860427856, |
| "sampling/importance_sampling_ratio/min": 0.33917105197906494, |
| "sampling/sampling_logp_difference/max": 1.444007158279419, |
| "sampling/sampling_logp_difference/mean": 0.033993300050497055, |
| "step": 146, |
| "step_time": 3.3651030065957457 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018057780805975198, |
| "clip_ratio/high_mean": 0.0018057780805975198, |
| "clip_ratio/low_mean": 0.00021574972197413445, |
| "clip_ratio/low_min": 0.00021574972197413445, |
| "clip_ratio/region_mean": 0.0020215277560055255, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 291.0, |
| "completions/max_terminated_length": 291.0, |
| "completions/mean_length": 108.83999633789062, |
| "completions/mean_terminated_length": 108.83999633789062, |
| "completions/min_length": 18.0, |
| "completions/min_terminated_length": 18.0, |
| "entropy": 0.22867086827754973, |
| "epoch": 0.39945652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00997087825089693, |
| "learning_rate": 5e-05, |
| "loss": -0.00039731746073812246, |
| "num_tokens": 6805621.0, |
| "reward": 0.8268554210662842, |
| "reward_std": 0.3209690749645233, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826074957847595, |
| "rewards/length_penalty/mean": -0.05314452946186066, |
| "rewards/length_penalty/std": 0.03853520378470421, |
| "sampling/importance_sampling_ratio/max": 2.606764078140259, |
| "sampling/importance_sampling_ratio/mean": 0.9894402027130127, |
| "sampling/importance_sampling_ratio/min": 0.3649539351463318, |
| "sampling/sampling_logp_difference/max": 1.0079841613769531, |
| "sampling/sampling_logp_difference/mean": 0.028062352910637856, |
| "step": 147, |
| "step_time": 4.092198684811592 |
| }, |
| { |
| "clip_ratio/high_max": 0.0017266561510041357, |
| "clip_ratio/high_mean": 0.0017266561510041357, |
| "clip_ratio/low_mean": 0.000488461391068995, |
| "clip_ratio/low_min": 0.000488461391068995, |
| "clip_ratio/region_mean": 0.0022151174955070017, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 661.0, |
| "completions/max_terminated_length": 661.0, |
| "completions/mean_length": 137.94000244140625, |
| "completions/mean_terminated_length": 137.94000244140625, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.2801614224910736, |
| "epoch": 0.40217391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013792288489639759, |
| "learning_rate": 5e-05, |
| "loss": 0.010532232001423836, |
| "num_tokens": 6815428.0, |
| "reward": 0.7726464867591858, |
| "reward_std": 0.3749605715274811, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.0673535168170929, |
| "rewards/length_penalty/std": 0.06557419151067734, |
| "sampling/importance_sampling_ratio/max": 2.05609130859375, |
| "sampling/importance_sampling_ratio/mean": 0.9886090159416199, |
| "sampling/importance_sampling_ratio/min": 0.36234790086746216, |
| "sampling/sampling_logp_difference/max": 1.015150547027588, |
| "sampling/sampling_logp_difference/mean": 0.03219727799296379, |
| "step": 148, |
| "step_time": 7.033994301920757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0022396529093384743, |
| "clip_ratio/high_mean": 0.0022396529093384743, |
| "clip_ratio/low_mean": 0.0004771304433234036, |
| "clip_ratio/low_min": 0.0004771304433234036, |
| "clip_ratio/region_mean": 0.002716783294454217, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 335.0, |
| "completions/max_terminated_length": 335.0, |
| "completions/mean_length": 168.97999572753906, |
| "completions/mean_terminated_length": 168.97999572753906, |
| "completions/min_length": 52.0, |
| "completions/min_terminated_length": 52.0, |
| "entropy": 0.2570392400026321, |
| "epoch": 0.4048913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009006854146718979, |
| "learning_rate": 5e-05, |
| "loss": 0.0003677508793771267, |
| "num_tokens": 6826847.0, |
| "reward": 0.2374902218580246, |
| "reward_std": 0.49559831619262695, |
| "rewards/correctness/mean": 0.3199999928474426, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.08250976353883743, |
| "rewards/length_penalty/std": 0.04074801132082939, |
| "sampling/importance_sampling_ratio/max": 2.1414754390716553, |
| "sampling/importance_sampling_ratio/mean": 0.9905763864517212, |
| "sampling/importance_sampling_ratio/min": 0.3560671806335449, |
| "sampling/sampling_logp_difference/max": 1.0326359272003174, |
| "sampling/sampling_logp_difference/mean": 0.028651099652051926, |
| "step": 149, |
| "step_time": 4.350436957087368 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008030201541259884, |
| "clip_ratio/high_mean": 0.0008030201541259884, |
| "clip_ratio/low_mean": 0.0008030201541259884, |
| "clip_ratio/low_min": 0.0008030201541259884, |
| "clip_ratio/region_mean": 0.0016060403082519769, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 110.0, |
| "completions/max_terminated_length": 110.0, |
| "completions/mean_length": 43.89999771118164, |
| "completions/mean_terminated_length": 43.89999771118164, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.17641887366771697, |
| "epoch": 0.4076086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008080766536295414, |
| "learning_rate": 5e-05, |
| "loss": -0.003685970790684223, |
| "num_tokens": 6830802.0, |
| "reward": 0.6985644102096558, |
| "reward_std": 0.4497761130332947, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.02143554762005806, |
| "rewards/length_penalty/std": 0.011126765981316566, |
| "sampling/importance_sampling_ratio/max": 1.8722953796386719, |
| "sampling/importance_sampling_ratio/mean": 0.993256151676178, |
| "sampling/importance_sampling_ratio/min": 0.3265569508075714, |
| "sampling/sampling_logp_difference/max": 1.1191508769989014, |
| "sampling/sampling_logp_difference/mean": 0.025421978905797005, |
| "step": 150, |
| "step_time": 1.9895036513917148 |
| }, |
| { |
| "clip_ratio/high_max": 0.001909730408806354, |
| "clip_ratio/high_mean": 0.001909730408806354, |
| "clip_ratio/low_mean": 0.0007019498269073665, |
| "clip_ratio/low_min": 0.0007019498269073665, |
| "clip_ratio/region_mean": 0.002611680258996785, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 450.0, |
| "completions/max_terminated_length": 450.0, |
| "completions/mean_length": 104.37999725341797, |
| "completions/mean_terminated_length": 104.37999725341797, |
| "completions/min_length": 33.0, |
| "completions/min_terminated_length": 33.0, |
| "entropy": 0.30302112698554995, |
| "epoch": 0.41032608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017708539962768555, |
| "learning_rate": 5e-05, |
| "loss": 0.01343751884996891, |
| "num_tokens": 6838991.0, |
| "reward": 0.8890331983566284, |
| "reward_std": 0.27069416642189026, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.05096679553389549, |
| "rewards/length_penalty/std": 0.045334592461586, |
| "sampling/importance_sampling_ratio/max": 2.2641215324401855, |
| "sampling/importance_sampling_ratio/mean": 0.9894137382507324, |
| "sampling/importance_sampling_ratio/min": 0.31814077496528625, |
| "sampling/sampling_logp_difference/max": 1.145261287689209, |
| "sampling/sampling_logp_difference/mean": 0.03505779057741165, |
| "step": 151, |
| "step_time": 4.9901159319560975 |
| }, |
| { |
| "clip_ratio/high_max": 0.0020658338442444803, |
| "clip_ratio/high_mean": 0.0020658338442444803, |
| "clip_ratio/low_mean": 0.00022172948811203242, |
| "clip_ratio/low_min": 0.00022172948811203242, |
| "clip_ratio/region_mean": 0.0022875633323565124, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 403.0, |
| "completions/max_terminated_length": 403.0, |
| "completions/mean_length": 69.95999908447266, |
| "completions/mean_terminated_length": 69.95999908447266, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.3296308219432831, |
| "epoch": 0.41304347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014323486015200615, |
| "learning_rate": 5e-05, |
| "loss": 0.0010965061374008656, |
| "num_tokens": 6845669.0, |
| "reward": 0.9258398413658142, |
| "reward_std": 0.210097536444664, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.034160155802965164, |
| "rewards/length_penalty/std": 0.03389851003885269, |
| "sampling/importance_sampling_ratio/max": 1.7374582290649414, |
| "sampling/importance_sampling_ratio/mean": 0.9878533482551575, |
| "sampling/importance_sampling_ratio/min": 0.2914775311946869, |
| "sampling/sampling_logp_difference/max": 1.2327923774719238, |
| "sampling/sampling_logp_difference/mean": 0.03526343032717705, |
| "step": 152, |
| "step_time": 4.637364404974505 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014145854220259935, |
| "clip_ratio/high_mean": 0.0014145854220259935, |
| "clip_ratio/low_mean": 5.602241144515574e-05, |
| "clip_ratio/low_min": 5.602241144515574e-05, |
| "clip_ratio/region_mean": 0.0014706078334711492, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1165.0, |
| "completions/max_terminated_length": 1165.0, |
| "completions/mean_length": 289.8800048828125, |
| "completions/mean_terminated_length": 289.8800048828125, |
| "completions/min_length": 54.0, |
| "completions/min_terminated_length": 54.0, |
| "entropy": 0.3570341467857361, |
| "epoch": 0.4157608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01623588055372238, |
| "learning_rate": 5e-05, |
| "loss": -0.01666785404086113, |
| "num_tokens": 6863683.0, |
| "reward": 0.3584570288658142, |
| "reward_std": 0.5585468411445618, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.1415429711341858, |
| "rewards/length_penalty/std": 0.12034020572900772, |
| "sampling/importance_sampling_ratio/max": 2.3237290382385254, |
| "sampling/importance_sampling_ratio/mean": 0.9866432547569275, |
| "sampling/importance_sampling_ratio/min": 0.452730268239975, |
| "sampling/sampling_logp_difference/max": 0.8431732654571533, |
| "sampling/sampling_logp_difference/mean": 0.03062153421342373, |
| "step": 153, |
| "step_time": 12.238479646854103 |
| }, |
| { |
| "clip_ratio/high_max": 0.0023650207556784155, |
| "clip_ratio/high_mean": 0.0023650207556784155, |
| "clip_ratio/low_mean": 0.00022675737272948027, |
| "clip_ratio/low_min": 0.00022675737272948027, |
| "clip_ratio/region_mean": 0.002591778105124831, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 201.0, |
| "completions/max_terminated_length": 201.0, |
| "completions/mean_length": 111.31999969482422, |
| "completions/mean_terminated_length": 111.31999969482422, |
| "completions/min_length": 45.0, |
| "completions/min_terminated_length": 45.0, |
| "entropy": 0.2378071665763855, |
| "epoch": 0.41847826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010363536886870861, |
| "learning_rate": 5e-05, |
| "loss": 0.0004417411983013153, |
| "num_tokens": 6873009.0, |
| "reward": 0.5056445002555847, |
| "reward_std": 0.5034895539283752, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.05435546860098839, |
| "rewards/length_penalty/std": 0.018973326310515404, |
| "sampling/importance_sampling_ratio/max": 1.7399930953979492, |
| "sampling/importance_sampling_ratio/mean": 0.9895638227462769, |
| "sampling/importance_sampling_ratio/min": 0.37739306688308716, |
| "sampling/sampling_logp_difference/max": 0.9744679927825928, |
| "sampling/sampling_logp_difference/mean": 0.029702970758080482, |
| "step": 154, |
| "step_time": 3.0129952027928084 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007251268136315048, |
| "clip_ratio/high_mean": 0.0007251268136315048, |
| "clip_ratio/low_mean": 0.0002612145501188934, |
| "clip_ratio/low_min": 0.0002612145501188934, |
| "clip_ratio/region_mean": 0.0009863413637503982, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2040.0, |
| "completions/mean_length": 466.3399963378906, |
| "completions/mean_terminated_length": 328.8043518066406, |
| "completions/min_length": 31.0, |
| "completions/min_terminated_length": 31.0, |
| "entropy": 0.3204962432384491, |
| "epoch": 0.421195652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02144881896674633, |
| "learning_rate": 5e-05, |
| "loss": 0.03292368724942207, |
| "num_tokens": 6899356.0, |
| "reward": 0.5122948884963989, |
| "reward_std": 0.6860557198524475, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.22770507633686066, |
| "rewards/length_penalty/std": 0.32957208156585693, |
| "sampling/importance_sampling_ratio/max": 2.477583646774292, |
| "sampling/importance_sampling_ratio/mean": 0.9873630404472351, |
| "sampling/importance_sampling_ratio/min": 0.145801842212677, |
| "sampling/sampling_logp_difference/max": 1.925506830215454, |
| "sampling/sampling_logp_difference/mean": 0.02497403509914875, |
| "step": 155, |
| "step_time": 22.262837967136875 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016966318944469094, |
| "clip_ratio/high_mean": 0.0016966318944469094, |
| "clip_ratio/low_mean": 0.0005891928565688431, |
| "clip_ratio/low_min": 0.0005891928565688431, |
| "clip_ratio/region_mean": 0.0022858247626572846, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1319.0, |
| "completions/mean_length": 268.7599792480469, |
| "completions/mean_terminated_length": 232.448974609375, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.31526569426059725, |
| "epoch": 0.42391304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024226536974310875, |
| "learning_rate": 5e-05, |
| "loss": 0.023074232041835785, |
| "num_tokens": 6916354.0, |
| "reward": 0.788769543170929, |
| "reward_std": 0.43313589692115784, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.13123047351837158, |
| "rewards/length_penalty/std": 0.21081988513469696, |
| "sampling/importance_sampling_ratio/max": 2.111845016479492, |
| "sampling/importance_sampling_ratio/mean": 0.9881788492202759, |
| "sampling/importance_sampling_ratio/min": 0.29661259055137634, |
| "sampling/sampling_logp_difference/max": 1.2153284549713135, |
| "sampling/sampling_logp_difference/mean": 0.027346931397914886, |
| "step": 156, |
| "step_time": 20.626690889243037 |
| }, |
| { |
| "clip_ratio/high_max": 0.0026269301772117613, |
| "clip_ratio/high_mean": 0.0026269301772117613, |
| "clip_ratio/low_mean": 0.0015538995387032628, |
| "clip_ratio/low_min": 0.0015538995387032628, |
| "clip_ratio/region_mean": 0.004180829832330346, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 239.0, |
| "completions/max_terminated_length": 239.0, |
| "completions/mean_length": 81.36000061035156, |
| "completions/mean_terminated_length": 81.36000061035156, |
| "completions/min_length": 16.0, |
| "completions/min_terminated_length": 16.0, |
| "entropy": 0.24512319564819335, |
| "epoch": 0.4266304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010789505206048489, |
| "learning_rate": 5e-05, |
| "loss": -0.0028433026745915413, |
| "num_tokens": 6923102.0, |
| "reward": 0.7402734160423279, |
| "reward_std": 0.4102163314819336, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.039726562798023224, |
| "rewards/length_penalty/std": 0.025849608704447746, |
| "sampling/importance_sampling_ratio/max": 2.7470545768737793, |
| "sampling/importance_sampling_ratio/mean": 0.990745484828949, |
| "sampling/importance_sampling_ratio/min": 0.45892199873924255, |
| "sampling/sampling_logp_difference/max": 1.0105292797088623, |
| "sampling/sampling_logp_difference/mean": 0.03431452810764313, |
| "step": 157, |
| "step_time": 3.1062621069140732 |
| }, |
| { |
| "clip_ratio/high_max": 0.002015978051349521, |
| "clip_ratio/high_mean": 0.002015978051349521, |
| "clip_ratio/low_mean": 0.00019930244889110327, |
| "clip_ratio/low_min": 0.00019930244889110327, |
| "clip_ratio/region_mean": 0.002215280500240624, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1362.0, |
| "completions/mean_length": 267.3800048828125, |
| "completions/mean_terminated_length": 193.1875, |
| "completions/min_length": 32.0, |
| "completions/min_terminated_length": 32.0, |
| "entropy": 0.3066037118434906, |
| "epoch": 0.42934782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019704675301909447, |
| "learning_rate": 5e-05, |
| "loss": 0.07820351421833038, |
| "num_tokens": 6939881.0, |
| "reward": 0.6894433498382568, |
| "reward_std": 0.5227276086807251, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.1305566430091858, |
| "rewards/length_penalty/std": 0.2184125930070877, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9874107837677002, |
| "sampling/importance_sampling_ratio/min": 0.4051153063774109, |
| "sampling/sampling_logp_difference/max": 1.1135804653167725, |
| "sampling/sampling_logp_difference/mean": 0.026231875643134117, |
| "step": 158, |
| "step_time": 21.139369979966432 |
| }, |
| { |
| "clip_ratio/high_max": 0.0036007760791108012, |
| "clip_ratio/high_mean": 0.0036007760791108012, |
| "clip_ratio/low_mean": 0.0005286535713821649, |
| "clip_ratio/low_min": 0.0005286535713821649, |
| "clip_ratio/region_mean": 0.00412942967377603, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 311.0, |
| "completions/max_terminated_length": 311.0, |
| "completions/mean_length": 93.13999938964844, |
| "completions/mean_terminated_length": 93.13999938964844, |
| "completions/min_length": 37.0, |
| "completions/min_terminated_length": 37.0, |
| "entropy": 0.270704647898674, |
| "epoch": 0.4320652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010844467207789421, |
| "learning_rate": 5e-05, |
| "loss": 0.0017642746679484844, |
| "num_tokens": 6947808.0, |
| "reward": 0.7945214509963989, |
| "reward_std": 0.38281598687171936, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.045478515326976776, |
| "rewards/length_penalty/std": 0.027892369776964188, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9885851144790649, |
| "sampling/importance_sampling_ratio/min": 0.35911431908607483, |
| "sampling/sampling_logp_difference/max": 1.1696937084197998, |
| "sampling/sampling_logp_difference/mean": 0.035336412489414215, |
| "step": 159, |
| "step_time": 3.856142339296639 |
| }, |
| { |
| "clip_ratio/high_max": 0.003056347987148911, |
| "clip_ratio/high_mean": 0.003056347987148911, |
| "clip_ratio/low_mean": 0.00012763241538777949, |
| "clip_ratio/low_min": 0.00012763241538777949, |
| "clip_ratio/region_mean": 0.003183980344329029, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 393.0, |
| "completions/max_terminated_length": 393.0, |
| "completions/mean_length": 176.17999267578125, |
| "completions/mean_terminated_length": 176.17999267578125, |
| "completions/min_length": 65.0, |
| "completions/min_terminated_length": 65.0, |
| "entropy": 0.2691596210002899, |
| "epoch": 0.43478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017808452248573303, |
| "learning_rate": 5e-05, |
| "loss": -8.661765605211258e-06, |
| "num_tokens": 6960167.0, |
| "reward": 0.8339745998382568, |
| "reward_std": 0.2837885022163391, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.08602538704872131, |
| "rewards/length_penalty/std": 0.03771448880434036, |
| "sampling/importance_sampling_ratio/max": 2.4753832817077637, |
| "sampling/importance_sampling_ratio/mean": 0.9890847206115723, |
| "sampling/importance_sampling_ratio/min": 0.25265780091285706, |
| "sampling/sampling_logp_difference/max": 1.3757193088531494, |
| "sampling/sampling_logp_difference/mean": 0.03175670653581619, |
| "step": 160, |
| "step_time": 4.759896617848426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018184274638770148, |
| "clip_ratio/high_mean": 0.0018184274638770148, |
| "clip_ratio/low_mean": 0.00037750662013422696, |
| "clip_ratio/low_min": 0.00037750662013422696, |
| "clip_ratio/region_mean": 0.0021959340840112416, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2028.0, |
| "completions/mean_length": 387.94000244140625, |
| "completions/mean_terminated_length": 281.9787292480469, |
| "completions/min_length": 21.0, |
| "completions/min_terminated_length": 21.0, |
| "entropy": 0.5686160206794739, |
| "epoch": 0.4375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012659218162298203, |
| "learning_rate": 5e-05, |
| "loss": 0.015315894037485123, |
| "num_tokens": 6982994.0, |
| "reward": 0.5105761885643005, |
| "reward_std": 0.7217123508453369, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.1894238293170929, |
| "rewards/length_penalty/std": 0.31975671648979187, |
| "sampling/importance_sampling_ratio/max": 2.0985939502716064, |
| "sampling/importance_sampling_ratio/mean": 0.9789088368415833, |
| "sampling/importance_sampling_ratio/min": 0.46538472175598145, |
| "sampling/sampling_logp_difference/max": 0.7648909091949463, |
| "sampling/sampling_logp_difference/mean": 0.04148508235812187, |
| "step": 161, |
| "step_time": 21.801013707881793 |
| }, |
| { |
| "clip_ratio/high_max": 0.0022810002556070685, |
| "clip_ratio/high_mean": 0.0022810002556070685, |
| "clip_ratio/low_mean": 0.00028503030771389606, |
| "clip_ratio/low_min": 0.00028503030771389606, |
| "clip_ratio/region_mean": 0.0025660306215286254, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 389.0, |
| "completions/max_terminated_length": 389.0, |
| "completions/mean_length": 143.3000030517578, |
| "completions/mean_terminated_length": 143.3000030517578, |
| "completions/min_length": 44.0, |
| "completions/min_terminated_length": 44.0, |
| "entropy": 0.23762179017066956, |
| "epoch": 0.44021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016811639070510864, |
| "learning_rate": 5e-05, |
| "loss": 0.01475286204367876, |
| "num_tokens": 6992749.0, |
| "reward": 0.6700292825698853, |
| "reward_std": 0.4359190762042999, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.06997070461511612, |
| "rewards/length_penalty/std": 0.03450389578938484, |
| "sampling/importance_sampling_ratio/max": 2.3417956829071045, |
| "sampling/importance_sampling_ratio/mean": 0.9900611639022827, |
| "sampling/importance_sampling_ratio/min": 0.37993767857551575, |
| "sampling/sampling_logp_difference/max": 0.9677480459213257, |
| "sampling/sampling_logp_difference/mean": 0.031710945069789886, |
| "step": 162, |
| "step_time": 4.475936716189608 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011258373968303203, |
| "clip_ratio/high_mean": 0.0011258373968303203, |
| "clip_ratio/low_mean": 0.00027894002851098777, |
| "clip_ratio/low_min": 0.00027894002851098777, |
| "clip_ratio/region_mean": 0.0014047774486243725, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 195.0, |
| "completions/max_terminated_length": 195.0, |
| "completions/mean_length": 74.33999633789062, |
| "completions/mean_terminated_length": 74.33999633789062, |
| "completions/min_length": 35.0, |
| "completions/min_terminated_length": 35.0, |
| "entropy": 0.227192023396492, |
| "epoch": 0.4429347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011993701569736004, |
| "learning_rate": 5e-05, |
| "loss": 0.00753851467743516, |
| "num_tokens": 6999176.0, |
| "reward": 0.9237011671066284, |
| "reward_std": 0.2046716809272766, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.03629882633686066, |
| "rewards/length_penalty/std": 0.012223804369568825, |
| "sampling/importance_sampling_ratio/max": 2.1342215538024902, |
| "sampling/importance_sampling_ratio/mean": 0.9901916980743408, |
| "sampling/importance_sampling_ratio/min": 0.3072393238544464, |
| "sampling/sampling_logp_difference/max": 1.1801283359527588, |
| "sampling/sampling_logp_difference/mean": 0.03727969899773598, |
| "step": 163, |
| "step_time": 3.2619427188765258 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014262479264289141, |
| "clip_ratio/high_mean": 0.0014262479264289141, |
| "clip_ratio/low_mean": 0.0003521514940075576, |
| "clip_ratio/low_min": 0.0003521514940075576, |
| "clip_ratio/region_mean": 0.0017783994087949395, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 260.0, |
| "completions/max_terminated_length": 260.0, |
| "completions/mean_length": 159.22000122070312, |
| "completions/mean_terminated_length": 159.22000122070312, |
| "completions/min_length": 34.0, |
| "completions/min_terminated_length": 34.0, |
| "entropy": 0.25208097100257876, |
| "epoch": 0.44565217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017162201926112175, |
| "learning_rate": 5e-05, |
| "loss": 0.009491069242358208, |
| "num_tokens": 7009997.0, |
| "reward": 0.5222558379173279, |
| "reward_std": 0.5117260813713074, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.07774414122104645, |
| "rewards/length_penalty/std": 0.030168350785970688, |
| "sampling/importance_sampling_ratio/max": 2.4342620372772217, |
| "sampling/importance_sampling_ratio/mean": 0.9892648458480835, |
| "sampling/importance_sampling_ratio/min": 0.3828932046890259, |
| "sampling/sampling_logp_difference/max": 0.9599992036819458, |
| "sampling/sampling_logp_difference/mean": 0.03160088509321213, |
| "step": 164, |
| "step_time": 3.4830891208257526 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007399938767775893, |
| "clip_ratio/high_mean": 0.0007399938767775893, |
| "clip_ratio/low_mean": 0.0008091844152659178, |
| "clip_ratio/low_min": 0.0008091844152659178, |
| "clip_ratio/region_mean": 0.0015491783386096358, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 155.0, |
| "completions/max_terminated_length": 155.0, |
| "completions/mean_length": 77.55999755859375, |
| "completions/mean_terminated_length": 77.55999755859375, |
| "completions/min_length": 37.0, |
| "completions/min_terminated_length": 37.0, |
| "entropy": 0.27121474146842955, |
| "epoch": 0.4483695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011080972850322723, |
| "learning_rate": 5e-05, |
| "loss": 0.00039395957719534636, |
| "num_tokens": 7017245.0, |
| "reward": 0.7421289086341858, |
| "reward_std": 0.4145321547985077, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.037871092557907104, |
| "rewards/length_penalty/std": 0.01567479409277439, |
| "sampling/importance_sampling_ratio/max": 1.981217861175537, |
| "sampling/importance_sampling_ratio/mean": 0.9892025589942932, |
| "sampling/importance_sampling_ratio/min": 0.32550525665283203, |
| "sampling/sampling_logp_difference/max": 1.1223766803741455, |
| "sampling/sampling_logp_difference/mean": 0.03402024507522583, |
| "step": 165, |
| "step_time": 2.505642694188282 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016038015950471164, |
| "clip_ratio/high_mean": 0.0016038015950471164, |
| "clip_ratio/low_mean": 0.000671140942722559, |
| "clip_ratio/low_min": 0.000671140942722559, |
| "clip_ratio/region_mean": 0.0022749425377696754, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 127.0, |
| "completions/max_terminated_length": 127.0, |
| "completions/mean_length": 49.29999923706055, |
| "completions/mean_terminated_length": 49.29999923706055, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.18421165347099305, |
| "epoch": 0.45108695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009308744221925735, |
| "learning_rate": 5e-05, |
| "loss": -0.0009645363898016512, |
| "num_tokens": 7021620.0, |
| "reward": 0.8359277248382568, |
| "reward_std": 0.3553500473499298, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.02407226525247097, |
| "rewards/length_penalty/std": 0.015800654888153076, |
| "sampling/importance_sampling_ratio/max": 2.426981210708618, |
| "sampling/importance_sampling_ratio/mean": 0.9945739507675171, |
| "sampling/importance_sampling_ratio/min": 0.47531673312187195, |
| "sampling/sampling_logp_difference/max": 0.8866481781005859, |
| "sampling/sampling_logp_difference/mean": 0.0298598725348711, |
| "step": 166, |
| "step_time": 2.1257548579014838 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018261599820107222, |
| "clip_ratio/high_mean": 0.0018261599820107222, |
| "clip_ratio/low_mean": 0.0008409326546825469, |
| "clip_ratio/low_min": 0.0008409326546825469, |
| "clip_ratio/region_mean": 0.0026670926774386315, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1004.0, |
| "completions/max_terminated_length": 1004.0, |
| "completions/mean_length": 194.739990234375, |
| "completions/mean_terminated_length": 194.739990234375, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.33175224661827085, |
| "epoch": 0.453804347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013448608107864857, |
| "learning_rate": 5e-05, |
| "loss": 0.003722331952303648, |
| "num_tokens": 7034777.0, |
| "reward": 0.4449121057987213, |
| "reward_std": 0.559876561164856, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.09508789330720901, |
| "rewards/length_penalty/std": 0.1051904633641243, |
| "sampling/importance_sampling_ratio/max": 2.497279405593872, |
| "sampling/importance_sampling_ratio/mean": 0.9887950420379639, |
| "sampling/importance_sampling_ratio/min": 0.3122704029083252, |
| "sampling/sampling_logp_difference/max": 1.1638858318328857, |
| "sampling/sampling_logp_difference/mean": 0.030904069542884827, |
| "step": 167, |
| "step_time": 10.566271085757762 |
| }, |
| { |
| "clip_ratio/high_max": 0.0027813970344141127, |
| "clip_ratio/high_mean": 0.0027813970344141127, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0027813970344141127, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 311.0, |
| "completions/max_terminated_length": 311.0, |
| "completions/mean_length": 82.29999542236328, |
| "completions/mean_terminated_length": 82.29999542236328, |
| "completions/min_length": 18.0, |
| "completions/min_terminated_length": 18.0, |
| "entropy": 0.2749664276838303, |
| "epoch": 0.45652173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004291764926165342, |
| "learning_rate": 5e-05, |
| "loss": 0.0014910745667293668, |
| "num_tokens": 7041732.0, |
| "reward": 0.7198144197463989, |
| "reward_std": 0.4380590319633484, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.04018554836511612, |
| "rewards/length_penalty/std": 0.033490896224975586, |
| "sampling/importance_sampling_ratio/max": 2.1294050216674805, |
| "sampling/importance_sampling_ratio/mean": 0.9865438342094421, |
| "sampling/importance_sampling_ratio/min": 0.32641488313674927, |
| "sampling/sampling_logp_difference/max": 1.1195859909057617, |
| "sampling/sampling_logp_difference/mean": 0.033342134207487106, |
| "step": 168, |
| "step_time": 3.8433888640720397 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015060298377647997, |
| "clip_ratio/high_mean": 0.0015060298377647997, |
| "clip_ratio/low_mean": 0.0014212609501555562, |
| "clip_ratio/low_min": 0.0014212609501555562, |
| "clip_ratio/region_mean": 0.0029272907646372913, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1250.0, |
| "completions/max_terminated_length": 1250.0, |
| "completions/mean_length": 146.67999267578125, |
| "completions/mean_terminated_length": 146.67999267578125, |
| "completions/min_length": 23.0, |
| "completions/min_terminated_length": 23.0, |
| "entropy": 0.33983248472213745, |
| "epoch": 0.4592391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007958698086440563, |
| "learning_rate": 5e-05, |
| "loss": 0.005089750047773123, |
| "num_tokens": 7052136.0, |
| "reward": 0.3083789050579071, |
| "reward_std": 0.5274860858917236, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.07162109017372131, |
| "rewards/length_penalty/std": 0.09010875225067139, |
| "sampling/importance_sampling_ratio/max": 2.429049015045166, |
| "sampling/importance_sampling_ratio/mean": 0.9872545003890991, |
| "sampling/importance_sampling_ratio/min": 0.25707486271858215, |
| "sampling/sampling_logp_difference/max": 1.3583879470825195, |
| "sampling/sampling_logp_difference/mean": 0.035425540059804916, |
| "step": 169, |
| "step_time": 12.240187409101054 |
| }, |
| { |
| "clip_ratio/high_max": 0.001748804305680096, |
| "clip_ratio/high_mean": 0.001748804305680096, |
| "clip_ratio/low_mean": 0.00028377799317240714, |
| "clip_ratio/low_min": 0.00028377799317240714, |
| "clip_ratio/region_mean": 0.0020325822988525033, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1551.0, |
| "completions/mean_length": 253.33999633789062, |
| "completions/mean_terminated_length": 138.7872314453125, |
| "completions/min_length": 45.0, |
| "completions/min_terminated_length": 45.0, |
| "entropy": 0.36185679137706755, |
| "epoch": 0.46195652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015123526565730572, |
| "learning_rate": 5e-05, |
| "loss": 0.06740963459014893, |
| "num_tokens": 7067053.0, |
| "reward": 0.456298828125, |
| "reward_std": 0.6356388330459595, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.12370117008686066, |
| "rewards/length_penalty/std": 0.2486318200826645, |
| "sampling/importance_sampling_ratio/max": 2.287450075149536, |
| "sampling/importance_sampling_ratio/mean": 0.9853586554527283, |
| "sampling/importance_sampling_ratio/min": 0.41997265815734863, |
| "sampling/sampling_logp_difference/max": 0.8675656318664551, |
| "sampling/sampling_logp_difference/mean": 0.03125310689210892, |
| "step": 170, |
| "step_time": 20.81000596913509 |
| }, |
| { |
| "clip_ratio/high_max": 0.0033981793094426394, |
| "clip_ratio/high_mean": 0.0033981793094426394, |
| "clip_ratio/low_mean": 0.00023255813866853713, |
| "clip_ratio/low_min": 0.00023255813866853713, |
| "clip_ratio/region_mean": 0.0036307374481111764, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 164.0, |
| "completions/max_terminated_length": 164.0, |
| "completions/mean_length": 74.47999572753906, |
| "completions/mean_terminated_length": 74.47999572753906, |
| "completions/min_length": 18.0, |
| "completions/min_terminated_length": 18.0, |
| "entropy": 0.2514496833086014, |
| "epoch": 0.46467391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011234855279326439, |
| "learning_rate": 5e-05, |
| "loss": -0.0006487057544291019, |
| "num_tokens": 7073547.0, |
| "reward": 0.8236327767372131, |
| "reward_std": 0.34646114706993103, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.03636718913912773, |
| "rewards/length_penalty/std": 0.018819812685251236, |
| "sampling/importance_sampling_ratio/max": 1.777944803237915, |
| "sampling/importance_sampling_ratio/mean": 0.9895418286323547, |
| "sampling/importance_sampling_ratio/min": 0.39342281222343445, |
| "sampling/sampling_logp_difference/max": 0.9328703880310059, |
| "sampling/sampling_logp_difference/mean": 0.02890367992222309, |
| "step": 171, |
| "step_time": 2.4440803304314613 |
| }, |
| { |
| "clip_ratio/high_max": 0.001812327210791409, |
| "clip_ratio/high_mean": 0.001812327210791409, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001812327210791409, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 264.0, |
| "completions/max_terminated_length": 264.0, |
| "completions/mean_length": 79.0, |
| "completions/mean_terminated_length": 79.0, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.2839715212583542, |
| "epoch": 0.4673913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011006252840161324, |
| "learning_rate": 5e-05, |
| "loss": 0.00581324053928256, |
| "num_tokens": 7080997.0, |
| "reward": 0.5614257454872131, |
| "reward_std": 0.5160092115402222, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.03857421875, |
| "rewards/length_penalty/std": 0.033149152994155884, |
| "sampling/importance_sampling_ratio/max": 2.571403980255127, |
| "sampling/importance_sampling_ratio/mean": 0.9876032471656799, |
| "sampling/importance_sampling_ratio/min": 0.32369109988212585, |
| "sampling/sampling_logp_difference/max": 1.1279656887054443, |
| "sampling/sampling_logp_difference/mean": 0.03325434774160385, |
| "step": 172, |
| "step_time": 3.586650517070666 |
| }, |
| { |
| "clip_ratio/high_max": 0.001602892787195742, |
| "clip_ratio/high_mean": 0.001602892787195742, |
| "clip_ratio/low_mean": 0.0004811125691048801, |
| "clip_ratio/low_min": 0.0004811125691048801, |
| "clip_ratio/region_mean": 0.0020840053679421545, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 728.0, |
| "completions/max_terminated_length": 728.0, |
| "completions/mean_length": 120.05999755859375, |
| "completions/mean_terminated_length": 120.05999755859375, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.43551568388938905, |
| "epoch": 0.4701086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011051046662032604, |
| "learning_rate": 5e-05, |
| "loss": 0.01577736996114254, |
| "num_tokens": 7090290.0, |
| "reward": 0.6013769507408142, |
| "reward_std": 0.5000688433647156, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.058623045682907104, |
| "rewards/length_penalty/std": 0.058581989258527756, |
| "sampling/importance_sampling_ratio/max": 1.9741122722625732, |
| "sampling/importance_sampling_ratio/mean": 0.9831089377403259, |
| "sampling/importance_sampling_ratio/min": 0.29803401231765747, |
| "sampling/sampling_logp_difference/max": 1.210547685623169, |
| "sampling/sampling_logp_difference/mean": 0.0391753688454628, |
| "step": 173, |
| "step_time": 7.968987070955336 |
| }, |
| { |
| "clip_ratio/high_max": 0.002187453955411911, |
| "clip_ratio/high_mean": 0.002187453955411911, |
| "clip_ratio/low_mean": 0.00030674845911562445, |
| "clip_ratio/low_min": 0.00030674845911562445, |
| "clip_ratio/region_mean": 0.0024942024145275356, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 124.0, |
| "completions/max_terminated_length": 124.0, |
| "completions/mean_length": 56.31999969482422, |
| "completions/mean_terminated_length": 56.31999969482422, |
| "completions/min_length": 12.0, |
| "completions/min_terminated_length": 12.0, |
| "entropy": 0.22374942600727082, |
| "epoch": 0.47282608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010718019679188728, |
| "learning_rate": 5e-05, |
| "loss": 0.003098592394962907, |
| "num_tokens": 7095346.0, |
| "reward": 0.7524999976158142, |
| "reward_std": 0.4143524467945099, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.027499999850988388, |
| "rewards/length_penalty/std": 0.01560207549482584, |
| "sampling/importance_sampling_ratio/max": 1.9330708980560303, |
| "sampling/importance_sampling_ratio/mean": 0.9878627061843872, |
| "sampling/importance_sampling_ratio/min": 0.3103266656398773, |
| "sampling/sampling_logp_difference/max": 1.1701297760009766, |
| "sampling/sampling_logp_difference/mean": 0.030313508585095406, |
| "step": 174, |
| "step_time": 2.133615422062576 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015244228998199105, |
| "clip_ratio/high_mean": 0.0015244228998199105, |
| "clip_ratio/low_mean": 0.00018198362085968255, |
| "clip_ratio/low_min": 0.00018198362085968255, |
| "clip_ratio/region_mean": 0.0017064065439626574, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 270.0, |
| "completions/max_terminated_length": 270.0, |
| "completions/mean_length": 99.07999420166016, |
| "completions/mean_terminated_length": 99.07999420166016, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.23095017969608306, |
| "epoch": 0.47554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009856120683252811, |
| "learning_rate": 5e-05, |
| "loss": -0.000948627945035696, |
| "num_tokens": 7103380.0, |
| "reward": 0.7316210865974426, |
| "reward_std": 0.41359657049179077, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.04837890714406967, |
| "rewards/length_penalty/std": 0.035482726991176605, |
| "sampling/importance_sampling_ratio/max": 2.0748867988586426, |
| "sampling/importance_sampling_ratio/mean": 0.9920177459716797, |
| "sampling/importance_sampling_ratio/min": 0.3595147132873535, |
| "sampling/sampling_logp_difference/max": 1.0230002403259277, |
| "sampling/sampling_logp_difference/mean": 0.02558811940252781, |
| "step": 175, |
| "step_time": 3.36774894525297 |
| }, |
| { |
| "clip_ratio/high_max": 0.004104181891307235, |
| "clip_ratio/high_mean": 0.004104181891307235, |
| "clip_ratio/low_mean": 0.00019665684085339307, |
| "clip_ratio/low_min": 0.00019665684085339307, |
| "clip_ratio/region_mean": 0.004300838755443692, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 334.0, |
| "completions/max_terminated_length": 334.0, |
| "completions/mean_length": 95.5999984741211, |
| "completions/mean_terminated_length": 95.5999984741211, |
| "completions/min_length": 21.0, |
| "completions/min_terminated_length": 21.0, |
| "entropy": 0.32383630871772767, |
| "epoch": 0.4782608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013959099538624287, |
| "learning_rate": 5e-05, |
| "loss": 0.0028878431767225266, |
| "num_tokens": 7110800.0, |
| "reward": 0.6733202934265137, |
| "reward_std": 0.44744816422462463, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.04667968675494194, |
| "rewards/length_penalty/std": 0.04310058057308197, |
| "sampling/importance_sampling_ratio/max": 2.884258985519409, |
| "sampling/importance_sampling_ratio/mean": 0.987141489982605, |
| "sampling/importance_sampling_ratio/min": 0.43793705105781555, |
| "sampling/sampling_logp_difference/max": 1.0592679977416992, |
| "sampling/sampling_logp_difference/mean": 0.03418336436152458, |
| "step": 176, |
| "step_time": 3.9611348991747946 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013552463613450527, |
| "clip_ratio/high_mean": 0.0013552463613450527, |
| "clip_ratio/low_mean": 0.00031298904214054345, |
| "clip_ratio/low_min": 0.00031298904214054345, |
| "clip_ratio/region_mean": 0.0016682354267686605, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 117.0, |
| "completions/max_terminated_length": 117.0, |
| "completions/mean_length": 58.91999816894531, |
| "completions/mean_terminated_length": 58.91999816894531, |
| "completions/min_length": 17.0, |
| "completions/min_terminated_length": 17.0, |
| "entropy": 0.22237029671669006, |
| "epoch": 0.48097826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00716294115409255, |
| "learning_rate": 5e-05, |
| "loss": -9.61346086114645e-05, |
| "num_tokens": 7115976.0, |
| "reward": 0.6912304759025574, |
| "reward_std": 0.4565470218658447, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573422908783, |
| "rewards/length_penalty/mean": -0.02876953035593033, |
| "rewards/length_penalty/std": 0.011297140270471573, |
| "sampling/importance_sampling_ratio/max": 2.5116207599639893, |
| "sampling/importance_sampling_ratio/mean": 0.9901822805404663, |
| "sampling/importance_sampling_ratio/min": 0.3565656840801239, |
| "sampling/sampling_logp_difference/max": 1.0312368869781494, |
| "sampling/sampling_logp_difference/mean": 0.029166536405682564, |
| "step": 177, |
| "step_time": 2.089208585675806 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007905138656497002, |
| "clip_ratio/high_mean": 0.0007905138656497002, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007905138656497002, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 104.0, |
| "completions/max_terminated_length": 104.0, |
| "completions/mean_length": 38.97999954223633, |
| "completions/mean_terminated_length": 38.97999954223633, |
| "completions/min_length": 13.0, |
| "completions/min_terminated_length": 13.0, |
| "entropy": 0.15681053400039674, |
| "epoch": 0.483695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011388145387172699, |
| "learning_rate": 5e-05, |
| "loss": -0.00011829048162326217, |
| "num_tokens": 7120385.0, |
| "reward": 0.7209667563438416, |
| "reward_std": 0.44683605432510376, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308748841285706, |
| "rewards/length_penalty/mean": -0.019033202901482582, |
| "rewards/length_penalty/std": 0.012724212370812893, |
| "sampling/importance_sampling_ratio/max": 1.9708325862884521, |
| "sampling/importance_sampling_ratio/mean": 0.9927529692649841, |
| "sampling/importance_sampling_ratio/min": 0.19267265498638153, |
| "sampling/sampling_logp_difference/max": 1.6467626094818115, |
| "sampling/sampling_logp_difference/mean": 0.02143104188144207, |
| "step": 178, |
| "step_time": 1.9621772288810462 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013067401945590974, |
| "clip_ratio/high_mean": 0.0013067401945590974, |
| "clip_ratio/low_mean": 0.0003149606287479401, |
| "clip_ratio/low_min": 0.0003149606287479401, |
| "clip_ratio/region_mean": 0.0016217008233070374, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 159.0, |
| "completions/max_terminated_length": 159.0, |
| "completions/mean_length": 61.18000030517578, |
| "completions/mean_terminated_length": 61.18000030517578, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.23652730882167816, |
| "epoch": 0.48641304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01019375305622816, |
| "learning_rate": 5e-05, |
| "loss": -0.0025170999579131603, |
| "num_tokens": 7126244.0, |
| "reward": 0.6501269340515137, |
| "reward_std": 0.4624166488647461, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.029873047024011612, |
| "rewards/length_penalty/std": 0.016003957018256187, |
| "sampling/importance_sampling_ratio/max": 2.436744451522827, |
| "sampling/importance_sampling_ratio/mean": 0.9908562302589417, |
| "sampling/importance_sampling_ratio/min": 0.4560236632823944, |
| "sampling/sampling_logp_difference/max": 0.8906629085540771, |
| "sampling/sampling_logp_difference/mean": 0.02767454832792282, |
| "step": 179, |
| "step_time": 2.3798739509657025 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015546044800430537, |
| "clip_ratio/high_mean": 0.0015546044800430537, |
| "clip_ratio/low_mean": 0.0005685532581992447, |
| "clip_ratio/low_min": 0.0005685532581992447, |
| "clip_ratio/region_mean": 0.002123157726600766, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2013.0, |
| "completions/mean_length": 303.41998291015625, |
| "completions/mean_terminated_length": 267.8163146972656, |
| "completions/min_length": 33.0, |
| "completions/min_terminated_length": 33.0, |
| "entropy": 0.4458068490028381, |
| "epoch": 0.4891304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008136949501931667, |
| "learning_rate": 5e-05, |
| "loss": 0.012830721214413643, |
| "num_tokens": 7145875.0, |
| "reward": 0.37184569239616394, |
| "reward_std": 0.6702902317047119, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.14815430343151093, |
| "rewards/length_penalty/std": 0.2554056644439697, |
| "sampling/importance_sampling_ratio/max": 2.0383386611938477, |
| "sampling/importance_sampling_ratio/mean": 0.9838020205497742, |
| "sampling/importance_sampling_ratio/min": 0.42360401153564453, |
| "sampling/sampling_logp_difference/max": 0.8589562177658081, |
| "sampling/sampling_logp_difference/mean": 0.032803554087877274, |
| "step": 180, |
| "step_time": 21.541535117896274 |
| }, |
| { |
| "clip_ratio/high_max": 0.0022305613267235456, |
| "clip_ratio/high_mean": 0.0022305613267235456, |
| "clip_ratio/low_mean": 0.0011865235341247172, |
| "clip_ratio/low_min": 0.0011865235341247172, |
| "clip_ratio/region_mean": 0.003417084884131327, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1500.0, |
| "completions/max_terminated_length": 1500.0, |
| "completions/mean_length": 192.8199920654297, |
| "completions/mean_terminated_length": 192.8199920654297, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.4294774532318115, |
| "epoch": 0.49184782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014891736209392548, |
| "learning_rate": 5e-05, |
| "loss": 0.0005076061934232712, |
| "num_tokens": 7158676.0, |
| "reward": 0.44584959745407104, |
| "reward_std": 0.5859835147857666, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.09415039420127869, |
| "rewards/length_penalty/std": 0.14089402556419373, |
| "sampling/importance_sampling_ratio/max": 2.572859287261963, |
| "sampling/importance_sampling_ratio/mean": 0.9847277998924255, |
| "sampling/importance_sampling_ratio/min": 0.4930843114852905, |
| "sampling/sampling_logp_difference/max": 0.9450178146362305, |
| "sampling/sampling_logp_difference/mean": 0.033227670937776566, |
| "step": 181, |
| "step_time": 14.99796702992171 |
| }, |
| { |
| "clip_ratio/high_max": 0.002261120337061584, |
| "clip_ratio/high_mean": 0.002261120337061584, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.002261120337061584, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 160.0, |
| "completions/max_terminated_length": 160.0, |
| "completions/mean_length": 73.36000061035156, |
| "completions/mean_terminated_length": 73.36000061035156, |
| "completions/min_length": 31.0, |
| "completions/min_terminated_length": 31.0, |
| "entropy": 0.31624099612236023, |
| "epoch": 0.4945652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01414932869374752, |
| "learning_rate": 5e-05, |
| "loss": -0.0010051056742668152, |
| "num_tokens": 7165184.0, |
| "reward": 0.464179664850235, |
| "reward_std": 0.5021290183067322, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.035820312798023224, |
| "rewards/length_penalty/std": 0.016128452494740486, |
| "sampling/importance_sampling_ratio/max": 2.4239728450775146, |
| "sampling/importance_sampling_ratio/mean": 0.988733172416687, |
| "sampling/importance_sampling_ratio/min": 0.21040107309818268, |
| "sampling/sampling_logp_difference/max": 1.5587396621704102, |
| "sampling/sampling_logp_difference/mean": 0.02917325124144554, |
| "step": 182, |
| "step_time": 2.445704023586586 |
| }, |
| { |
| "clip_ratio/high_max": 0.0024747584015130998, |
| "clip_ratio/high_mean": 0.0024747584015130998, |
| "clip_ratio/low_mean": 0.0011320129502564668, |
| "clip_ratio/low_min": 0.0011320129502564668, |
| "clip_ratio/region_mean": 0.0036067713517695665, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 153.0, |
| "completions/max_terminated_length": 153.0, |
| "completions/mean_length": 58.19999694824219, |
| "completions/mean_terminated_length": 58.19999694824219, |
| "completions/min_length": 21.0, |
| "completions/min_terminated_length": 21.0, |
| "entropy": 0.2677886724472046, |
| "epoch": 0.49728260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012314711697399616, |
| "learning_rate": 5e-05, |
| "loss": 0.001457427628338337, |
| "num_tokens": 7170384.0, |
| "reward": 0.9115819931030273, |
| "reward_std": 0.2446342408657074, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.02841796912252903, |
| "rewards/length_penalty/std": 0.015301529318094254, |
| "sampling/importance_sampling_ratio/max": 1.5758779048919678, |
| "sampling/importance_sampling_ratio/mean": 0.9906679391860962, |
| "sampling/importance_sampling_ratio/min": 0.3942355811595917, |
| "sampling/sampling_logp_difference/max": 0.9308066368103027, |
| "sampling/sampling_logp_difference/mean": 0.026903966441750526, |
| "step": 183, |
| "step_time": 2.842646149918437 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010377714177593588, |
| "clip_ratio/high_mean": 0.0010377714177593588, |
| "clip_ratio/low_mean": 0.0010451125097461044, |
| "clip_ratio/low_min": 0.0010451125097461044, |
| "clip_ratio/region_mean": 0.0020828839275054633, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 433.0, |
| "completions/max_terminated_length": 433.0, |
| "completions/mean_length": 93.57999420166016, |
| "completions/mean_terminated_length": 93.57999420166016, |
| "completions/min_length": 16.0, |
| "completions/min_terminated_length": 16.0, |
| "entropy": 0.37457822561264037, |
| "epoch": 0.5, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010905577801167965, |
| "learning_rate": 5e-05, |
| "loss": -0.0019602077081799507, |
| "num_tokens": 7177813.0, |
| "reward": 0.5543066263198853, |
| "reward_std": 0.5123629570007324, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.04569336026906967, |
| "rewards/length_penalty/std": 0.05604249984025955, |
| "sampling/importance_sampling_ratio/max": 1.7604007720947266, |
| "sampling/importance_sampling_ratio/mean": 0.9877779483795166, |
| "sampling/importance_sampling_ratio/min": 0.20862749218940735, |
| "sampling/sampling_logp_difference/max": 1.5672049522399902, |
| "sampling/sampling_logp_difference/mean": 0.0326719805598259, |
| "step": 184, |
| "step_time": 4.801921327831224 |
| }, |
| { |
| "clip_ratio/high_max": 0.0025969252397771924, |
| "clip_ratio/high_mean": 0.0025969252397771924, |
| "clip_ratio/low_mean": 0.0005546173313632607, |
| "clip_ratio/low_min": 0.0005546173313632607, |
| "clip_ratio/region_mean": 0.003151542547857389, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 995.0, |
| "completions/mean_length": 172.16000366210938, |
| "completions/mean_terminated_length": 133.87754821777344, |
| "completions/min_length": 23.0, |
| "completions/min_terminated_length": 23.0, |
| "entropy": 0.4305099666118622, |
| "epoch": 0.5027173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009968320839107037, |
| "learning_rate": 5e-05, |
| "loss": 0.005090097431093454, |
| "num_tokens": 7190141.0, |
| "reward": 0.25593748688697815, |
| "reward_std": 0.5455606579780579, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.08406250178813934, |
| "rewards/length_penalty/std": 0.16681182384490967, |
| "sampling/importance_sampling_ratio/max": 1.7391831874847412, |
| "sampling/importance_sampling_ratio/mean": 0.9834895133972168, |
| "sampling/importance_sampling_ratio/min": 0.30923932790756226, |
| "sampling/sampling_logp_difference/max": 1.1736397743225098, |
| "sampling/sampling_logp_difference/mean": 0.03487389162182808, |
| "step": 185, |
| "step_time": 20.135311206802726 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011485486757010221, |
| "clip_ratio/high_mean": 0.0011485486757010221, |
| "clip_ratio/low_mean": 0.0003176651312969625, |
| "clip_ratio/low_min": 0.0003176651312969625, |
| "clip_ratio/region_mean": 0.001466213818639517, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 721.0, |
| "completions/max_terminated_length": 721.0, |
| "completions/mean_length": 102.93999481201172, |
| "completions/mean_terminated_length": 102.93999481201172, |
| "completions/min_length": 28.0, |
| "completions/min_terminated_length": 28.0, |
| "entropy": 0.32348001599311826, |
| "epoch": 0.5054347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01584494300186634, |
| "learning_rate": 5e-05, |
| "loss": 0.022778289392590523, |
| "num_tokens": 7199358.0, |
| "reward": 0.2897363305091858, |
| "reward_std": 0.4857187867164612, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.050263673067092896, |
| "rewards/length_penalty/std": 0.05547233670949936, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9891101717948914, |
| "sampling/importance_sampling_ratio/min": 0.44916167855262756, |
| "sampling/sampling_logp_difference/max": 1.228992223739624, |
| "sampling/sampling_logp_difference/mean": 0.027644626796245575, |
| "step": 186, |
| "step_time": 7.652905150782317 |
| }, |
| { |
| "clip_ratio/high_max": 0.002942303172312677, |
| "clip_ratio/high_mean": 0.002942303172312677, |
| "clip_ratio/low_mean": 0.00043316339142620566, |
| "clip_ratio/low_min": 0.00043316339142620566, |
| "clip_ratio/region_mean": 0.0033754665637388825, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 142.0, |
| "completions/max_terminated_length": 142.0, |
| "completions/mean_length": 87.47999572753906, |
| "completions/mean_terminated_length": 87.47999572753906, |
| "completions/min_length": 40.0, |
| "completions/min_terminated_length": 40.0, |
| "entropy": 0.346056205034256, |
| "epoch": 0.5081521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009617852047085762, |
| "learning_rate": 5e-05, |
| "loss": -0.0005349450511857867, |
| "num_tokens": 7206792.0, |
| "reward": 0.7772851586341858, |
| "reward_std": 0.3886815309524536, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.04271484538912773, |
| "rewards/length_penalty/std": 0.013727720826864243, |
| "sampling/importance_sampling_ratio/max": 1.7080680131912231, |
| "sampling/importance_sampling_ratio/mean": 0.9851519465446472, |
| "sampling/importance_sampling_ratio/min": 0.3232705295085907, |
| "sampling/sampling_logp_difference/max": 1.1292657852172852, |
| "sampling/sampling_logp_difference/mean": 0.03272521123290062, |
| "step": 187, |
| "step_time": 2.3949871559161693 |
| }, |
| { |
| "clip_ratio/high_max": 0.003067309851758182, |
| "clip_ratio/high_mean": 0.003067309851758182, |
| "clip_ratio/low_mean": 0.000710010714828968, |
| "clip_ratio/low_min": 0.000710010714828968, |
| "clip_ratio/region_mean": 0.0037773205665871503, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 156.0, |
| "completions/max_terminated_length": 156.0, |
| "completions/mean_length": 59.23999786376953, |
| "completions/mean_terminated_length": 59.23999786376953, |
| "completions/min_length": 35.0, |
| "completions/min_terminated_length": 35.0, |
| "entropy": 0.2870894134044647, |
| "epoch": 0.5108695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00894978828728199, |
| "learning_rate": 5e-05, |
| "loss": 0.0006205665413290262, |
| "num_tokens": 7213674.0, |
| "reward": 0.9110742211341858, |
| "reward_std": 0.2406059205532074, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.028925782069563866, |
| "rewards/length_penalty/std": 0.012459544464945793, |
| "sampling/importance_sampling_ratio/max": 1.5915205478668213, |
| "sampling/importance_sampling_ratio/mean": 0.9866249561309814, |
| "sampling/importance_sampling_ratio/min": 0.25376468896865845, |
| "sampling/sampling_logp_difference/max": 1.3713479042053223, |
| "sampling/sampling_logp_difference/mean": 0.03077450767159462, |
| "step": 188, |
| "step_time": 2.506129484856501 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009468309581279755, |
| "clip_ratio/high_mean": 0.0009468309581279755, |
| "clip_ratio/low_mean": 0.000961519981501624, |
| "clip_ratio/low_min": 0.000961519981501624, |
| "clip_ratio/region_mean": 0.0019083509396295995, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1086.0, |
| "completions/max_terminated_length": 1086.0, |
| "completions/mean_length": 161.05999755859375, |
| "completions/mean_terminated_length": 161.05999755859375, |
| "completions/min_length": 40.0, |
| "completions/min_terminated_length": 40.0, |
| "entropy": 0.4137052595615387, |
| "epoch": 0.5135869565217391, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01786559261381626, |
| "learning_rate": 5e-05, |
| "loss": 0.010921262204647064, |
| "num_tokens": 7224827.0, |
| "reward": 0.6213573813438416, |
| "reward_std": 0.5171093344688416, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.0786425769329071, |
| "rewards/length_penalty/std": 0.09781408309936523, |
| "sampling/importance_sampling_ratio/max": 2.220214605331421, |
| "sampling/importance_sampling_ratio/mean": 0.9850063323974609, |
| "sampling/importance_sampling_ratio/min": 0.4292641282081604, |
| "sampling/sampling_logp_difference/max": 0.8456828594207764, |
| "sampling/sampling_logp_difference/mean": 0.03254811838269234, |
| "step": 189, |
| "step_time": 10.982258481439203 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009298557182773947, |
| "clip_ratio/high_mean": 0.0009298557182773947, |
| "clip_ratio/low_mean": 0.0004873778438195586, |
| "clip_ratio/low_min": 0.0004873778438195586, |
| "clip_ratio/region_mean": 0.0014172335620969533, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 229.0, |
| "completions/max_terminated_length": 229.0, |
| "completions/mean_length": 70.22000122070312, |
| "completions/mean_terminated_length": 70.22000122070312, |
| "completions/min_length": 24.0, |
| "completions/min_terminated_length": 24.0, |
| "entropy": 0.3256260991096497, |
| "epoch": 0.5163043478260869, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012585996650159359, |
| "learning_rate": 5e-05, |
| "loss": -0.0024468661285936832, |
| "num_tokens": 7230898.0, |
| "reward": 0.9057128429412842, |
| "reward_std": 0.24925492703914642, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.03428710997104645, |
| "rewards/length_penalty/std": 0.026151560246944427, |
| "sampling/importance_sampling_ratio/max": 2.1814732551574707, |
| "sampling/importance_sampling_ratio/mean": 0.9877171516418457, |
| "sampling/importance_sampling_ratio/min": 0.3333365023136139, |
| "sampling/sampling_logp_difference/max": 1.0986027717590332, |
| "sampling/sampling_logp_difference/mean": 0.03380505368113518, |
| "step": 190, |
| "step_time": 2.8829450951889157 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009009991772472858, |
| "clip_ratio/high_mean": 0.0009009991772472858, |
| "clip_ratio/low_mean": 0.0006575917359441519, |
| "clip_ratio/low_min": 0.0006575917359441519, |
| "clip_ratio/region_mean": 0.0015585909131914377, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 276.0, |
| "completions/max_terminated_length": 276.0, |
| "completions/mean_length": 63.65999984741211, |
| "completions/mean_terminated_length": 63.65999984741211, |
| "completions/min_length": 24.0, |
| "completions/min_terminated_length": 24.0, |
| "entropy": 0.36922687888145445, |
| "epoch": 0.5190217391304348, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010491478256881237, |
| "learning_rate": 5e-05, |
| "loss": 0.0018182226922363043, |
| "num_tokens": 7236391.0, |
| "reward": 0.6689159870147705, |
| "reward_std": 0.4714803695678711, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.031083984300494194, |
| "rewards/length_penalty/std": 0.021737802773714066, |
| "sampling/importance_sampling_ratio/max": 1.7441283464431763, |
| "sampling/importance_sampling_ratio/mean": 0.9856583476066589, |
| "sampling/importance_sampling_ratio/min": 0.4908250570297241, |
| "sampling/sampling_logp_difference/max": 0.711667537689209, |
| "sampling/sampling_logp_difference/mean": 0.03220402076840401, |
| "step": 191, |
| "step_time": 3.274767436552793 |
| }, |
| { |
| "clip_ratio/high_max": 0.002678904612548649, |
| "clip_ratio/high_mean": 0.002678904612548649, |
| "clip_ratio/low_mean": 0.0006638401304371655, |
| "clip_ratio/low_min": 0.0006638401304371655, |
| "clip_ratio/region_mean": 0.003342744731344283, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 908.0, |
| "completions/max_terminated_length": 908.0, |
| "completions/mean_length": 234.77999877929688, |
| "completions/mean_terminated_length": 234.77999877929688, |
| "completions/min_length": 33.0, |
| "completions/min_terminated_length": 33.0, |
| "entropy": 0.6806320786476135, |
| "epoch": 0.5217391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02616410329937935, |
| "learning_rate": 5e-05, |
| "loss": 0.05776366591453552, |
| "num_tokens": 7252930.0, |
| "reward": 0.28536131978034973, |
| "reward_std": 0.5705822110176086, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.11463867127895355, |
| "rewards/length_penalty/std": 0.10842815786600113, |
| "sampling/importance_sampling_ratio/max": 2.1739437580108643, |
| "sampling/importance_sampling_ratio/mean": 0.9770089387893677, |
| "sampling/importance_sampling_ratio/min": 0.4234622120857239, |
| "sampling/sampling_logp_difference/max": 0.8592910766601562, |
| "sampling/sampling_logp_difference/mean": 0.047108348459005356, |
| "step": 192, |
| "step_time": 10.75727180717513 |
| }, |
| { |
| "clip_ratio/high_max": 0.002673549111932516, |
| "clip_ratio/high_mean": 0.002673549111932516, |
| "clip_ratio/low_mean": 0.0003072196617722511, |
| "clip_ratio/low_min": 0.0003072196617722511, |
| "clip_ratio/region_mean": 0.0029807687737047673, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 257.0, |
| "completions/max_terminated_length": 257.0, |
| "completions/mean_length": 82.04000091552734, |
| "completions/mean_terminated_length": 82.04000091552734, |
| "completions/min_length": 15.0, |
| "completions/min_terminated_length": 15.0, |
| "entropy": 0.27706671953201295, |
| "epoch": 0.5244565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011028271168470383, |
| "learning_rate": 5e-05, |
| "loss": -0.0024434439837932587, |
| "num_tokens": 7259162.0, |
| "reward": 0.5199413895606995, |
| "reward_std": 0.48865923285484314, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.040058594197034836, |
| "rewards/length_penalty/std": 0.0338875986635685, |
| "sampling/importance_sampling_ratio/max": 1.7838475704193115, |
| "sampling/importance_sampling_ratio/mean": 0.9892287850379944, |
| "sampling/importance_sampling_ratio/min": 0.5288249254226685, |
| "sampling/sampling_logp_difference/max": 0.6370978355407715, |
| "sampling/sampling_logp_difference/mean": 0.02674698270857334, |
| "step": 193, |
| "step_time": 3.1811520259361714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0021951367147266866, |
| "clip_ratio/high_mean": 0.0021951367147266866, |
| "clip_ratio/low_mean": 0.0002810163889080286, |
| "clip_ratio/low_min": 0.0002810163889080286, |
| "clip_ratio/region_mean": 0.0024761531269177793, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2022.0, |
| "completions/max_terminated_length": 2022.0, |
| "completions/mean_length": 158.59999084472656, |
| "completions/mean_terminated_length": 158.59999084472656, |
| "completions/min_length": 18.0, |
| "completions/min_terminated_length": 18.0, |
| "entropy": 0.5259371221065521, |
| "epoch": 0.5271739130434783, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010996954515576363, |
| "learning_rate": 5e-05, |
| "loss": 0.010928381234407425, |
| "num_tokens": 7270752.0, |
| "reward": 0.6625585556030273, |
| "reward_std": 0.5541573762893677, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.07744140923023224, |
| "rewards/length_penalty/std": 0.18147680163383484, |
| "sampling/importance_sampling_ratio/max": 2.160527467727661, |
| "sampling/importance_sampling_ratio/mean": 0.9812043309211731, |
| "sampling/importance_sampling_ratio/min": 0.28289860486984253, |
| "sampling/sampling_logp_difference/max": 1.2626667022705078, |
| "sampling/sampling_logp_difference/mean": 0.03715597093105316, |
| "step": 194, |
| "step_time": 20.461127671878785 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013174308696761727, |
| "clip_ratio/high_mean": 0.0013174308696761727, |
| "clip_ratio/low_mean": 0.00045662098564207553, |
| "clip_ratio/low_min": 0.00045662098564207553, |
| "clip_ratio/region_mean": 0.0017740518553182483, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 180.0, |
| "completions/max_terminated_length": 180.0, |
| "completions/mean_length": 44.91999816894531, |
| "completions/mean_terminated_length": 44.91999816894531, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.2546180605888367, |
| "epoch": 0.529891304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009743832051753998, |
| "learning_rate": 5e-05, |
| "loss": 0.0027442362625151873, |
| "num_tokens": 7275358.0, |
| "reward": 0.43806639313697815, |
| "reward_std": 0.5021190643310547, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.02193359285593033, |
| "rewards/length_penalty/std": 0.01121544186025858, |
| "sampling/importance_sampling_ratio/max": 1.587057113647461, |
| "sampling/importance_sampling_ratio/mean": 0.9913366436958313, |
| "sampling/importance_sampling_ratio/min": 0.3795216381549835, |
| "sampling/sampling_logp_difference/max": 0.9688436985015869, |
| "sampling/sampling_logp_difference/mean": 0.025467142462730408, |
| "step": 195, |
| "step_time": 2.3768158222083002 |
| }, |
| { |
| "clip_ratio/high_max": 0.0024600768461823463, |
| "clip_ratio/high_mean": 0.0024600768461823463, |
| "clip_ratio/low_mean": 0.0005848227883689106, |
| "clip_ratio/low_min": 0.0005848227883689106, |
| "clip_ratio/region_mean": 0.0030448996927589177, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 607.0, |
| "completions/max_terminated_length": 607.0, |
| "completions/mean_length": 90.0999984741211, |
| "completions/mean_terminated_length": 90.0999984741211, |
| "completions/min_length": 38.0, |
| "completions/min_terminated_length": 38.0, |
| "entropy": 0.30158002972602843, |
| "epoch": 0.532608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008184398524463177, |
| "learning_rate": 5e-05, |
| "loss": 0.006111910566687584, |
| "num_tokens": 7282333.0, |
| "reward": 0.5360058546066284, |
| "reward_std": 0.5229967832565308, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.04399413987994194, |
| "rewards/length_penalty/std": 0.044464435428380966, |
| "sampling/importance_sampling_ratio/max": 2.022139072418213, |
| "sampling/importance_sampling_ratio/mean": 0.9878635406494141, |
| "sampling/importance_sampling_ratio/min": 0.3107838034629822, |
| "sampling/sampling_logp_difference/max": 1.1686577796936035, |
| "sampling/sampling_logp_difference/mean": 0.025973588228225708, |
| "step": 196, |
| "step_time": 6.1462403137702495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0029944060021080076, |
| "clip_ratio/high_mean": 0.0029944060021080076, |
| "clip_ratio/low_mean": 0.0010095724370330571, |
| "clip_ratio/low_min": 0.0010095724370330571, |
| "clip_ratio/region_mean": 0.004003978427499532, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1020.0, |
| "completions/mean_length": 198.09999084472656, |
| "completions/mean_terminated_length": 160.34693908691406, |
| "completions/min_length": 41.0, |
| "completions/min_terminated_length": 41.0, |
| "entropy": 0.5559770226478576, |
| "epoch": 0.5353260869565217, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020512336865067482, |
| "learning_rate": 5e-05, |
| "loss": 0.023510821163654327, |
| "num_tokens": 7295638.0, |
| "reward": 0.7032714486122131, |
| "reward_std": 0.4924466013908386, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.09672851860523224, |
| "rewards/length_penalty/std": 0.15379859507083893, |
| "sampling/importance_sampling_ratio/max": 2.211470603942871, |
| "sampling/importance_sampling_ratio/mean": 0.9787412285804749, |
| "sampling/importance_sampling_ratio/min": 0.3842313289642334, |
| "sampling/sampling_logp_difference/max": 0.9565105438232422, |
| "sampling/sampling_logp_difference/mean": 0.041975509375333786, |
| "step": 197, |
| "step_time": 20.05243187956512 |
| }, |
| { |
| "clip_ratio/high_max": 0.0028542080195620655, |
| "clip_ratio/high_mean": 0.0028542080195620655, |
| "clip_ratio/low_mean": 0.0012986538466066122, |
| "clip_ratio/low_min": 0.0012986538466066122, |
| "clip_ratio/region_mean": 0.004152861842885614, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 194.0, |
| "completions/max_terminated_length": 194.0, |
| "completions/mean_length": 57.0, |
| "completions/mean_terminated_length": 57.0, |
| "completions/min_length": 23.0, |
| "completions/min_terminated_length": 23.0, |
| "entropy": 0.36579219698905946, |
| "epoch": 0.5380434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010308896191418171, |
| "learning_rate": 5e-05, |
| "loss": 0.0019076343160122633, |
| "num_tokens": 7304398.0, |
| "reward": 0.7521679401397705, |
| "reward_std": 0.4286014437675476, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.02783203125, |
| "rewards/length_penalty/std": 0.016324035823345184, |
| "sampling/importance_sampling_ratio/max": 1.9980320930480957, |
| "sampling/importance_sampling_ratio/mean": 0.9866833090782166, |
| "sampling/importance_sampling_ratio/min": 0.3350236117839813, |
| "sampling/sampling_logp_difference/max": 1.0935542583465576, |
| "sampling/sampling_logp_difference/mean": 0.03588447719812393, |
| "step": 198, |
| "step_time": 3.3153429948724806 |
| }, |
| { |
| "clip_ratio/high_max": 0.000424628471955657, |
| "clip_ratio/high_mean": 0.000424628471955657, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000424628471955657, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 94.0, |
| "completions/max_terminated_length": 94.0, |
| "completions/mean_length": 45.5, |
| "completions/mean_terminated_length": 45.5, |
| "completions/min_length": 19.0, |
| "completions/min_terminated_length": 19.0, |
| "entropy": 0.28086246848106383, |
| "epoch": 0.5407608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0075567313469946384, |
| "learning_rate": 5e-05, |
| "loss": -0.0015296062920242548, |
| "num_tokens": 7309363.0, |
| "reward": 0.9577831625938416, |
| "reward_std": 0.14229898154735565, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.022216796875, |
| "rewards/length_penalty/std": 0.009976229630410671, |
| "sampling/importance_sampling_ratio/max": 1.7690128087997437, |
| "sampling/importance_sampling_ratio/mean": 0.9884426593780518, |
| "sampling/importance_sampling_ratio/min": 0.4450662434101105, |
| "sampling/sampling_logp_difference/max": 0.8095321655273438, |
| "sampling/sampling_logp_difference/mean": 0.02728041261434555, |
| "step": 199, |
| "step_time": 1.9000479853712022 |
| }, |
| { |
| "clip_ratio/high_max": 0.004609737906139344, |
| "clip_ratio/high_mean": 0.004609737906139344, |
| "clip_ratio/low_mean": 0.0008183412137441337, |
| "clip_ratio/low_min": 0.0008183412137441337, |
| "clip_ratio/region_mean": 0.005428079073317349, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1824.0, |
| "completions/max_terminated_length": 1824.0, |
| "completions/mean_length": 334.32000732421875, |
| "completions/mean_terminated_length": 334.32000732421875, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.8804279923439026, |
| "epoch": 0.5434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01084327232092619, |
| "learning_rate": 5e-05, |
| "loss": 0.016516633331775665, |
| "num_tokens": 7330339.0, |
| "reward": 0.1767578125, |
| "reward_std": 0.6074655652046204, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.1632421910762787, |
| "rewards/length_penalty/std": 0.2124619483947754, |
| "sampling/importance_sampling_ratio/max": 2.3384251594543457, |
| "sampling/importance_sampling_ratio/mean": 0.9735147953033447, |
| "sampling/importance_sampling_ratio/min": 0.31894412636756897, |
| "sampling/sampling_logp_difference/max": 1.1427392959594727, |
| "sampling/sampling_logp_difference/mean": 0.04999689757823944, |
| "step": 200, |
| "step_time": 19.11996822594665 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 200, |
| "num_input_tokens_seen": 7330339, |
| "num_train_epochs": 1, |
| "save_steps": 50, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|