| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.4076086956521739, |
| "eval_steps": 500, |
| "global_step": 150, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2017.0, |
| "completions/mean_length": 1671.2799072265625, |
| "completions/mean_terminated_length": 1398.4827880859375, |
| "completions/min_length": 992.0, |
| "completions/min_terminated_length": 992.0, |
| "entropy": 0.21248915791511536, |
| "epoch": 0.002717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01607886701822281, |
| "learning_rate": 0.0, |
| "loss": 0.07359933853149414, |
| "num_tokens": 86304.0, |
| "reward": -0.21605467796325684, |
| "reward_std": 0.656523585319519, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.8160547018051147, |
| "rewards/length_penalty/std": 0.18964654207229614, |
| "sampling/importance_sampling_ratio/max": 1.5222556591033936, |
| "sampling/importance_sampling_ratio/mean": 0.9927070140838623, |
| "sampling/importance_sampling_ratio/min": 0.6435883045196533, |
| "sampling/sampling_logp_difference/max": 0.44069600105285645, |
| "sampling/sampling_logp_difference/mean": 0.013988969847559929, |
| "step": 1, |
| "step_time": 27.394495805958286 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.6800000071525574, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1991.0, |
| "completions/mean_length": 1941.47998046875, |
| "completions/mean_terminated_length": 1715.125, |
| "completions/min_length": 1079.0, |
| "completions/min_terminated_length": 1079.0, |
| "entropy": 0.2905632793903351, |
| "epoch": 0.005434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01730405166745186, |
| "learning_rate": 5e-06, |
| "loss": 0.07159668952226639, |
| "num_tokens": 186618.0, |
| "reward": -0.7279882431030273, |
| "reward_std": 0.4917677640914917, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.9479882717132568, |
| "rewards/length_penalty/std": 0.10518474876880646, |
| "sampling/importance_sampling_ratio/max": 1.450933814048767, |
| "sampling/importance_sampling_ratio/mean": 0.9901063442230225, |
| "sampling/importance_sampling_ratio/min": 0.3688630163669586, |
| "sampling/sampling_logp_difference/max": 0.9973299503326416, |
| "sampling/sampling_logp_difference/mean": 0.01780563034117222, |
| "step": 2, |
| "step_time": 28.991049223113805 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009276511613279581, |
| "clip_ratio/high_mean": 0.0009276511613279581, |
| "clip_ratio/low_mean": 0.00012337134103290737, |
| "clip_ratio/low_min": 0.00012337134103290737, |
| "clip_ratio/region_mean": 0.0010510225081816315, |
| "completions/clipped_ratio": 0.3999999761581421, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2048.0, |
| "completions/mean_length": 1714.760009765625, |
| "completions/mean_terminated_length": 1492.60009765625, |
| "completions/min_length": 753.0, |
| "completions/min_terminated_length": 753.0, |
| "entropy": 0.2737257957458496, |
| "epoch": 0.008152173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017563248053193092, |
| "learning_rate": 1e-05, |
| "loss": 0.09476927667856216, |
| "num_tokens": 274626.0, |
| "reward": -0.21728515625, |
| "reward_std": 0.6381246447563171, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.8372851610183716, |
| "rewards/length_penalty/std": 0.20151396095752716, |
| "sampling/importance_sampling_ratio/max": 1.426476001739502, |
| "sampling/importance_sampling_ratio/mean": 0.9904518723487854, |
| "sampling/importance_sampling_ratio/min": 0.6474350690841675, |
| "sampling/sampling_logp_difference/max": 0.4347367286682129, |
| "sampling/sampling_logp_difference/mean": 0.01698637194931507, |
| "step": 3, |
| "step_time": 27.53572188084945 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008016214531380683, |
| "clip_ratio/high_mean": 0.0008016214531380683, |
| "clip_ratio/low_mean": 0.00010628659802023322, |
| "clip_ratio/low_min": 0.00010628659802023322, |
| "clip_ratio/region_mean": 0.0009079080424271524, |
| "completions/clipped_ratio": 0.3199999928474426, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1991.0, |
| "completions/mean_length": 1506.5399169921875, |
| "completions/mean_terminated_length": 1251.7353515625, |
| "completions/min_length": 638.0, |
| "completions/min_terminated_length": 638.0, |
| "entropy": 0.25913241803646087, |
| "epoch": 0.010869565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017625654116272926, |
| "learning_rate": 1.5e-05, |
| "loss": 0.12047722935676575, |
| "num_tokens": 352853.0, |
| "reward": -0.0356152318418026, |
| "reward_std": 0.6583986878395081, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.7356152534484863, |
| "rewards/length_penalty/std": 0.23944713175296783, |
| "sampling/importance_sampling_ratio/max": 1.4886895418167114, |
| "sampling/importance_sampling_ratio/mean": 0.9911117553710938, |
| "sampling/importance_sampling_ratio/min": 0.6022539734840393, |
| "sampling/sampling_logp_difference/max": 0.5070760250091553, |
| "sampling/sampling_logp_difference/mean": 0.016714798286557198, |
| "step": 4, |
| "step_time": 27.26718003093265 |
| }, |
| { |
| "clip_ratio/high_max": 0.00031515124428551646, |
| "clip_ratio/high_mean": 0.00031515124428551646, |
| "clip_ratio/low_mean": 0.0001696181825536769, |
| "clip_ratio/low_min": 0.0001696181825536769, |
| "clip_ratio/region_mean": 0.00048476942392881026, |
| "completions/clipped_ratio": 0.7599999904632568, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2022.0, |
| "completions/mean_length": 1914.3599853515625, |
| "completions/mean_terminated_length": 1491.166748046875, |
| "completions/min_length": 1072.0, |
| "completions/min_terminated_length": 1072.0, |
| "entropy": 0.2717311054468155, |
| "epoch": 0.01358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018083127215504646, |
| "learning_rate": 2e-05, |
| "loss": 0.08188383281230927, |
| "num_tokens": 452281.0, |
| "reward": -0.6947460770606995, |
| "reward_std": 0.5523213148117065, |
| "rewards/correctness/mean": 0.23999999463558197, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.9347460865974426, |
| "rewards/length_penalty/std": 0.13313990831375122, |
| "sampling/importance_sampling_ratio/max": 1.7996419668197632, |
| "sampling/importance_sampling_ratio/mean": 0.9906188249588013, |
| "sampling/importance_sampling_ratio/min": 0.6196915507316589, |
| "sampling/sampling_logp_difference/max": 0.5875877141952515, |
| "sampling/sampling_logp_difference/mean": 0.01734403893351555, |
| "step": 5, |
| "step_time": 28.98220992088318 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012173810973763466, |
| "clip_ratio/high_mean": 0.0012173810973763466, |
| "clip_ratio/low_mean": 8.390831935685128e-05, |
| "clip_ratio/low_min": 8.390831935685128e-05, |
| "clip_ratio/region_mean": 0.0013012894080020488, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2042.0, |
| "completions/mean_length": 1670.679931640625, |
| "completions/mean_terminated_length": 1397.4482421875, |
| "completions/min_length": 764.0, |
| "completions/min_terminated_length": 764.0, |
| "entropy": 0.25660490691661836, |
| "epoch": 0.016304347826086956, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016991887241601944, |
| "learning_rate": 2.5e-05, |
| "loss": 0.07579805701971054, |
| "num_tokens": 538805.0, |
| "reward": -0.23576171696186066, |
| "reward_std": 0.6744439005851746, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.8157617449760437, |
| "rewards/length_penalty/std": 0.21996504068374634, |
| "sampling/importance_sampling_ratio/max": 1.4670546054840088, |
| "sampling/importance_sampling_ratio/mean": 0.9912019371986389, |
| "sampling/importance_sampling_ratio/min": 0.6486796736717224, |
| "sampling/sampling_logp_difference/max": 0.4328162670135498, |
| "sampling/sampling_logp_difference/mean": 0.01650330238044262, |
| "step": 6, |
| "step_time": 27.390100880060345 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008734886534512043, |
| "clip_ratio/high_mean": 0.0008734886534512043, |
| "clip_ratio/low_mean": 6.886005430715159e-05, |
| "clip_ratio/low_min": 6.886005430715159e-05, |
| "clip_ratio/region_mean": 0.0009423487004823983, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1985.0, |
| "completions/mean_length": 1602.5799560546875, |
| "completions/mean_terminated_length": 1411.6856689453125, |
| "completions/min_length": 862.0, |
| "completions/min_terminated_length": 862.0, |
| "entropy": 0.24657512605190277, |
| "epoch": 0.019021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018574461340904236, |
| "learning_rate": 3e-05, |
| "loss": 0.10817548632621765, |
| "num_tokens": 621164.0, |
| "reward": -0.08250976353883743, |
| "reward_std": 0.6237230896949768, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.7825097441673279, |
| "rewards/length_penalty/std": 0.2039300501346588, |
| "sampling/importance_sampling_ratio/max": 1.899301528930664, |
| "sampling/importance_sampling_ratio/mean": 0.9912979006767273, |
| "sampling/importance_sampling_ratio/min": 0.5936238765716553, |
| "sampling/sampling_logp_difference/max": 0.6414861679077148, |
| "sampling/sampling_logp_difference/mean": 0.015865743160247803, |
| "step": 7, |
| "step_time": 27.102628064341843 |
| }, |
| { |
| "clip_ratio/high_max": 0.00030049889464862646, |
| "clip_ratio/high_mean": 0.00030049889464862646, |
| "clip_ratio/low_mean": 0.0001223457460582722, |
| "clip_ratio/low_min": 0.0001223457460582722, |
| "clip_ratio/region_mean": 0.00042284463997930286, |
| "completions/clipped_ratio": 0.5399999618530273, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2021.0, |
| "completions/mean_length": 1799.3199462890625, |
| "completions/mean_terminated_length": 1507.391357421875, |
| "completions/min_length": 766.0, |
| "completions/min_terminated_length": 766.0, |
| "entropy": 0.21842886209487916, |
| "epoch": 0.021739130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01700979843735695, |
| "learning_rate": 3.5e-05, |
| "loss": 0.07161665707826614, |
| "num_tokens": 713280.0, |
| "reward": -0.4185742139816284, |
| "reward_std": 0.6441221833229065, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8785741925239563, |
| "rewards/length_penalty/std": 0.16615615785121918, |
| "sampling/importance_sampling_ratio/max": 1.6265782117843628, |
| "sampling/importance_sampling_ratio/mean": 0.9922990202903748, |
| "sampling/importance_sampling_ratio/min": 0.634637713432312, |
| "sampling/sampling_logp_difference/max": 0.4864785671234131, |
| "sampling/sampling_logp_difference/mean": 0.014363318681716919, |
| "step": 8, |
| "step_time": 28.30564660113305 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009418363450095057, |
| "clip_ratio/high_mean": 0.0009418363450095057, |
| "clip_ratio/low_mean": 9.628144616726785e-05, |
| "clip_ratio/low_min": 9.628144616726785e-05, |
| "clip_ratio/region_mean": 0.0010381177882663906, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2048.0, |
| "completions/mean_length": 1622.97998046875, |
| "completions/mean_terminated_length": 1503.1025390625, |
| "completions/min_length": 873.0, |
| "completions/min_terminated_length": 873.0, |
| "entropy": 0.22798832952976228, |
| "epoch": 0.024456521739130436, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018526818603277206, |
| "learning_rate": 4e-05, |
| "loss": 0.11752811819314957, |
| "num_tokens": 796899.0, |
| "reward": -0.1524706929922104, |
| "reward_std": 0.5888569355010986, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732054233551, |
| "rewards/length_penalty/mean": -0.7924706935882568, |
| "rewards/length_penalty/std": 0.19514665007591248, |
| "sampling/importance_sampling_ratio/max": 1.4198806285858154, |
| "sampling/importance_sampling_ratio/mean": 0.9919716119766235, |
| "sampling/importance_sampling_ratio/min": 0.511864960193634, |
| "sampling/sampling_logp_difference/max": 0.6696943640708923, |
| "sampling/sampling_logp_difference/mean": 0.015113740228116512, |
| "step": 9, |
| "step_time": 27.674709191778675 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008836857683490962, |
| "clip_ratio/high_mean": 0.0008836857683490962, |
| "clip_ratio/low_mean": 0.0001610520586837083, |
| "clip_ratio/low_min": 0.0001610520586837083, |
| "clip_ratio/region_mean": 0.0010447378270328044, |
| "completions/clipped_ratio": 0.2800000011920929, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1945.0, |
| "completions/mean_length": 1527.5599365234375, |
| "completions/mean_terminated_length": 1325.1666259765625, |
| "completions/min_length": 650.0, |
| "completions/min_terminated_length": 650.0, |
| "entropy": 0.24462299942970275, |
| "epoch": 0.02717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017739074304699898, |
| "learning_rate": 4.5e-05, |
| "loss": 0.11460381746292114, |
| "num_tokens": 878097.0, |
| "reward": -0.0858789011836052, |
| "reward_std": 0.651462733745575, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.7458789348602295, |
| "rewards/length_penalty/std": 0.22174151241779327, |
| "sampling/importance_sampling_ratio/max": 1.4572726488113403, |
| "sampling/importance_sampling_ratio/mean": 0.9913933277130127, |
| "sampling/importance_sampling_ratio/min": 0.5709614157676697, |
| "sampling/sampling_logp_difference/max": 0.5604336261749268, |
| "sampling/sampling_logp_difference/mean": 0.01590959168970585, |
| "step": 10, |
| "step_time": 27.32456138706766 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005791342351585627, |
| "clip_ratio/high_mean": 0.0005791342351585627, |
| "clip_ratio/low_mean": 0.00020766998059116304, |
| "clip_ratio/low_min": 0.00020766998059116304, |
| "clip_ratio/region_mean": 0.0007868041866458952, |
| "completions/clipped_ratio": 0.5799999833106995, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2005.0, |
| "completions/mean_length": 1906.219970703125, |
| "completions/mean_terminated_length": 1710.4285888671875, |
| "completions/min_length": 1323.0, |
| "completions/min_terminated_length": 1323.0, |
| "entropy": 0.2523061394691467, |
| "epoch": 0.029891304347826088, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017356909811496735, |
| "learning_rate": 5e-05, |
| "loss": 0.074811190366745, |
| "num_tokens": 975828.0, |
| "reward": -0.4507714807987213, |
| "reward_std": 0.5823943614959717, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.9307714700698853, |
| "rewards/length_penalty/std": 0.1049875020980835, |
| "sampling/importance_sampling_ratio/max": 1.9958995580673218, |
| "sampling/importance_sampling_ratio/mean": 0.9913272857666016, |
| "sampling/importance_sampling_ratio/min": 0.6561847925186157, |
| "sampling/sampling_logp_difference/max": 0.6910948753356934, |
| "sampling/sampling_logp_difference/mean": 0.015796277672052383, |
| "step": 11, |
| "step_time": 28.954787808004767 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004949859721818939, |
| "clip_ratio/high_mean": 0.0004949859721818939, |
| "clip_ratio/low_mean": 0.0001681178982835263, |
| "clip_ratio/low_min": 0.0001681178982835263, |
| "clip_ratio/region_mean": 0.000663103861734271, |
| "completions/clipped_ratio": 0.4599999785423279, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1984.0, |
| "completions/mean_length": 1742.43994140625, |
| "completions/mean_terminated_length": 1482.148193359375, |
| "completions/min_length": 974.0, |
| "completions/min_terminated_length": 974.0, |
| "entropy": 0.2790819376707077, |
| "epoch": 0.03260869565217391, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017921727150678635, |
| "learning_rate": 5e-05, |
| "loss": 0.12824325263500214, |
| "num_tokens": 1065590.0, |
| "reward": -0.3508007824420929, |
| "reward_std": 0.6496888995170593, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.8508007526397705, |
| "rewards/length_penalty/std": 0.1730204075574875, |
| "sampling/importance_sampling_ratio/max": 1.4403636455535889, |
| "sampling/importance_sampling_ratio/mean": 0.9905098676681519, |
| "sampling/importance_sampling_ratio/min": 0.5996456146240234, |
| "sampling/sampling_logp_difference/max": 0.5114164352416992, |
| "sampling/sampling_logp_difference/mean": 0.01758064143359661, |
| "step": 12, |
| "step_time": 27.76311965705827 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006719783181324601, |
| "clip_ratio/high_mean": 0.0006719783181324601, |
| "clip_ratio/low_mean": 0.00014902167167747394, |
| "clip_ratio/low_min": 0.00014902167167747394, |
| "clip_ratio/region_mean": 0.0008209999883547425, |
| "completions/clipped_ratio": 0.4399999976158142, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2033.0, |
| "completions/mean_length": 1722.1400146484375, |
| "completions/mean_terminated_length": 1466.107177734375, |
| "completions/min_length": 936.0, |
| "completions/min_terminated_length": 936.0, |
| "entropy": 0.26495963633060454, |
| "epoch": 0.035326086956521736, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01803506910800934, |
| "learning_rate": 5e-05, |
| "loss": 0.10340610891580582, |
| "num_tokens": 1154717.0, |
| "reward": -0.3808886706829071, |
| "reward_std": 0.6363792419433594, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.8408886790275574, |
| "rewards/length_penalty/std": 0.17392964661121368, |
| "sampling/importance_sampling_ratio/max": 1.6443123817443848, |
| "sampling/importance_sampling_ratio/mean": 0.9907426238059998, |
| "sampling/importance_sampling_ratio/min": 0.6476600170135498, |
| "sampling/sampling_logp_difference/max": 0.49732232093811035, |
| "sampling/sampling_logp_difference/mean": 0.01705038733780384, |
| "step": 13, |
| "step_time": 27.700532860821113 |
| }, |
| { |
| "clip_ratio/high_max": 0.000790283753303811, |
| "clip_ratio/high_mean": 0.000790283753303811, |
| "clip_ratio/low_mean": 0.00015664642269257455, |
| "clip_ratio/low_min": 0.00015664642269257455, |
| "clip_ratio/region_mean": 0.0009469301789067685, |
| "completions/clipped_ratio": 0.4399999976158142, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1947.0, |
| "completions/mean_length": 1707.39990234375, |
| "completions/mean_terminated_length": 1439.7857666015625, |
| "completions/min_length": 939.0, |
| "completions/min_terminated_length": 939.0, |
| "entropy": 0.25884707272052765, |
| "epoch": 0.03804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01831369660794735, |
| "learning_rate": 5e-05, |
| "loss": 0.10717891156673431, |
| "num_tokens": 1242647.0, |
| "reward": -0.2536914050579071, |
| "reward_std": 0.6496968865394592, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.4985693693161011, |
| "rewards/length_penalty/mean": -0.833691418170929, |
| "rewards/length_penalty/std": 0.17602378129959106, |
| "sampling/importance_sampling_ratio/max": 1.4137200117111206, |
| "sampling/importance_sampling_ratio/mean": 0.9911543130874634, |
| "sampling/importance_sampling_ratio/min": 0.6645229458808899, |
| "sampling/sampling_logp_difference/max": 0.40868592262268066, |
| "sampling/sampling_logp_difference/mean": 0.016555489972233772, |
| "step": 14, |
| "step_time": 28.091995016904548 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007948186947032809, |
| "clip_ratio/high_mean": 0.0007948186947032809, |
| "clip_ratio/low_mean": 6.400095226126723e-05, |
| "clip_ratio/low_min": 6.400095226126723e-05, |
| "clip_ratio/region_mean": 0.0008588196593336761, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2024.0, |
| "completions/mean_length": 1566.659912109375, |
| "completions/mean_terminated_length": 1360.3714599609375, |
| "completions/min_length": 923.0, |
| "completions/min_terminated_length": 923.0, |
| "entropy": 0.19630298018455505, |
| "epoch": 0.04076086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01908237673342228, |
| "learning_rate": 5e-05, |
| "loss": 0.10991621017456055, |
| "num_tokens": 1323560.0, |
| "reward": -0.2849707007408142, |
| "reward_std": 0.6005774140357971, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.7649707198143005, |
| "rewards/length_penalty/std": 0.1954003870487213, |
| "sampling/importance_sampling_ratio/max": 1.4597245454788208, |
| "sampling/importance_sampling_ratio/mean": 0.9932765960693359, |
| "sampling/importance_sampling_ratio/min": 0.48890236020088196, |
| "sampling/sampling_logp_difference/max": 0.7155925035476685, |
| "sampling/sampling_logp_difference/mean": 0.013154168613255024, |
| "step": 15, |
| "step_time": 27.622011815896258 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007680640854232478, |
| "clip_ratio/high_mean": 0.0007680640854232478, |
| "clip_ratio/low_mean": 0.00010441521517350339, |
| "clip_ratio/low_min": 0.00010441521517350339, |
| "clip_ratio/region_mean": 0.0008724792875000276, |
| "completions/clipped_ratio": 0.2800000011920929, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2038.0, |
| "completions/mean_length": 1529.47998046875, |
| "completions/mean_terminated_length": 1327.8333740234375, |
| "completions/min_length": 648.0, |
| "completions/min_terminated_length": 648.0, |
| "entropy": 0.2527110666036606, |
| "epoch": 0.043478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019366132095456123, |
| "learning_rate": 5e-05, |
| "loss": 0.1180194765329361, |
| "num_tokens": 1404134.0, |
| "reward": -0.02681640535593033, |
| "reward_std": 0.6313645839691162, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.7468163967132568, |
| "rewards/length_penalty/std": 0.21961458027362823, |
| "sampling/importance_sampling_ratio/max": 1.4531008005142212, |
| "sampling/importance_sampling_ratio/mean": 0.990881621837616, |
| "sampling/importance_sampling_ratio/min": 0.45519256591796875, |
| "sampling/sampling_logp_difference/max": 0.7870347499847412, |
| "sampling/sampling_logp_difference/mean": 0.016263339668512344, |
| "step": 16, |
| "step_time": 26.84932530601509 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012978371232748032, |
| "clip_ratio/high_mean": 0.0012978371232748032, |
| "clip_ratio/low_mean": 4.327131027821451e-05, |
| "clip_ratio/low_min": 4.327131027821451e-05, |
| "clip_ratio/region_mean": 0.001341108442284167, |
| "completions/clipped_ratio": 0.5600000023841858, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2007.0, |
| "completions/mean_length": 1768.919921875, |
| "completions/mean_terminated_length": 1413.727294921875, |
| "completions/min_length": 865.0, |
| "completions/min_terminated_length": 865.0, |
| "entropy": 0.23181203305721282, |
| "epoch": 0.04619565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016713930293917656, |
| "learning_rate": 5e-05, |
| "loss": 0.08152732998132706, |
| "num_tokens": 1495230.0, |
| "reward": -0.4237304627895355, |
| "reward_std": 0.6704393029212952, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.8637304902076721, |
| "rewards/length_penalty/std": 0.20572726428508759, |
| "sampling/importance_sampling_ratio/max": 1.561582088470459, |
| "sampling/importance_sampling_ratio/mean": 0.9919421076774597, |
| "sampling/importance_sampling_ratio/min": 0.543276309967041, |
| "sampling/sampling_logp_difference/max": 0.6101372241973877, |
| "sampling/sampling_logp_difference/mean": 0.015333480201661587, |
| "step": 17, |
| "step_time": 27.873049326473847 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006598436157219112, |
| "clip_ratio/high_mean": 0.0006598436157219112, |
| "clip_ratio/low_mean": 0.00010337726125726476, |
| "clip_ratio/low_min": 0.00010337726125726476, |
| "clip_ratio/region_mean": 0.0007632208871655166, |
| "completions/clipped_ratio": 0.35999998450279236, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2017.0, |
| "completions/mean_length": 1746.0799560546875, |
| "completions/mean_terminated_length": 1576.25, |
| "completions/min_length": 1028.0, |
| "completions/min_terminated_length": 1028.0, |
| "entropy": 0.21664705574512483, |
| "epoch": 0.04891304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018009476363658905, |
| "learning_rate": 5e-05, |
| "loss": 0.08006907999515533, |
| "num_tokens": 1585034.0, |
| "reward": -0.2725781202316284, |
| "reward_std": 0.625697135925293, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.8525781035423279, |
| "rewards/length_penalty/std": 0.15941378474235535, |
| "sampling/importance_sampling_ratio/max": 1.639607310295105, |
| "sampling/importance_sampling_ratio/mean": 0.9925068020820618, |
| "sampling/importance_sampling_ratio/min": 0.4967416226863861, |
| "sampling/sampling_logp_difference/max": 0.6996852159500122, |
| "sampling/sampling_logp_difference/mean": 0.013985401019454002, |
| "step": 18, |
| "step_time": 28.311645316891372 |
| }, |
| { |
| "clip_ratio/high_max": 0.000741867849137634, |
| "clip_ratio/high_mean": 0.000741867849137634, |
| "clip_ratio/low_mean": 0.00018152535776607692, |
| "clip_ratio/low_min": 0.00018152535776607692, |
| "clip_ratio/region_mean": 0.0009233932243660093, |
| "completions/clipped_ratio": 0.35999998450279236, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2011.0, |
| "completions/mean_length": 1686.260009765625, |
| "completions/mean_terminated_length": 1482.78125, |
| "completions/min_length": 740.0, |
| "completions/min_terminated_length": 740.0, |
| "entropy": 0.2879431128501892, |
| "epoch": 0.051630434782608696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019337492063641548, |
| "learning_rate": 5e-05, |
| "loss": 0.08359130471944809, |
| "num_tokens": 1672777.0, |
| "reward": -0.18336912989616394, |
| "reward_std": 0.6349277496337891, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.8233691453933716, |
| "rewards/length_penalty/std": 0.19560563564300537, |
| "sampling/importance_sampling_ratio/max": 1.4635835886001587, |
| "sampling/importance_sampling_ratio/mean": 0.98971027135849, |
| "sampling/importance_sampling_ratio/min": 0.635256290435791, |
| "sampling/sampling_logp_difference/max": 0.45372676849365234, |
| "sampling/sampling_logp_difference/mean": 0.018413560464978218, |
| "step": 19, |
| "step_time": 27.62987573328428 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006679994985461235, |
| "clip_ratio/high_mean": 0.0006679994985461235, |
| "clip_ratio/low_mean": 5.8021536096930505e-05, |
| "clip_ratio/low_min": 5.8021536096930505e-05, |
| "clip_ratio/region_mean": 0.0007260210230015218, |
| "completions/clipped_ratio": 0.41999998688697815, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2046.0, |
| "completions/mean_length": 1787.9599609375, |
| "completions/mean_terminated_length": 1599.6551513671875, |
| "completions/min_length": 775.0, |
| "completions/min_terminated_length": 775.0, |
| "entropy": 0.2634012043476105, |
| "epoch": 0.05434782608695652, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018007466569542885, |
| "learning_rate": 5e-05, |
| "loss": 0.10140140354633331, |
| "num_tokens": 1764765.0, |
| "reward": -0.3130273222923279, |
| "reward_std": 0.627247154712677, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.8730273246765137, |
| "rewards/length_penalty/std": 0.17675596475601196, |
| "sampling/importance_sampling_ratio/max": 1.6090019941329956, |
| "sampling/importance_sampling_ratio/mean": 0.9909850358963013, |
| "sampling/importance_sampling_ratio/min": 0.6370789408683777, |
| "sampling/sampling_logp_difference/max": 0.475614070892334, |
| "sampling/sampling_logp_difference/mean": 0.01683841086924076, |
| "step": 20, |
| "step_time": 28.34779040887952 |
| }, |
| { |
| "clip_ratio/high_max": 0.000976240006275475, |
| "clip_ratio/high_mean": 0.000976240006275475, |
| "clip_ratio/low_mean": 7.026160965324379e-05, |
| "clip_ratio/low_min": 7.026160965324379e-05, |
| "clip_ratio/region_mean": 0.00104650161229074, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 1709.5399169921875, |
| "completions/mean_terminated_length": 1564.4857177734375, |
| "completions/min_length": 1061.0, |
| "completions/min_terminated_length": 1061.0, |
| "entropy": 0.20998104512691498, |
| "epoch": 0.057065217391304345, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018870066851377487, |
| "learning_rate": 5e-05, |
| "loss": 0.08603541553020477, |
| "num_tokens": 1853652.0, |
| "reward": -0.1347363293170929, |
| "reward_std": 0.5705064535140991, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.8347363471984863, |
| "rewards/length_penalty/std": 0.15160603821277618, |
| "sampling/importance_sampling_ratio/max": 1.4565383195877075, |
| "sampling/importance_sampling_ratio/mean": 0.9924175143241882, |
| "sampling/importance_sampling_ratio/min": 0.6045414805412292, |
| "sampling/sampling_logp_difference/max": 0.5032849311828613, |
| "sampling/sampling_logp_difference/mean": 0.01400977373123169, |
| "step": 21, |
| "step_time": 28.186939923092723 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007286557811312377, |
| "clip_ratio/high_mean": 0.0007286557811312377, |
| "clip_ratio/low_mean": 5.019558957428671e-05, |
| "clip_ratio/low_min": 5.019558957428671e-05, |
| "clip_ratio/region_mean": 0.0007788513612467796, |
| "completions/clipped_ratio": 0.29999998211860657, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2031.0, |
| "completions/mean_length": 1543.7799072265625, |
| "completions/mean_terminated_length": 1327.6856689453125, |
| "completions/min_length": 758.0, |
| "completions/min_terminated_length": 758.0, |
| "entropy": 0.2528064101934433, |
| "epoch": 0.059782608695652176, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019696656614542007, |
| "learning_rate": 5e-05, |
| "loss": 0.10337188839912415, |
| "num_tokens": 1935911.0, |
| "reward": -0.05379882827401161, |
| "reward_std": 0.6398226022720337, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.7537988424301147, |
| "rewards/length_penalty/std": 0.21059876680374146, |
| "sampling/importance_sampling_ratio/max": 2.2368545532226562, |
| "sampling/importance_sampling_ratio/mean": 0.9912877082824707, |
| "sampling/importance_sampling_ratio/min": 0.6480669975280762, |
| "sampling/sampling_logp_difference/max": 0.8050706386566162, |
| "sampling/sampling_logp_difference/mean": 0.016178736463189125, |
| "step": 22, |
| "step_time": 27.681587701197714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008125284919515252, |
| "clip_ratio/high_mean": 0.0008125284919515252, |
| "clip_ratio/low_mean": 0.00017936505464604123, |
| "clip_ratio/low_min": 0.00017936505464604123, |
| "clip_ratio/region_mean": 0.000991893548052758, |
| "completions/clipped_ratio": 0.3400000035762787, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2040.0, |
| "completions/mean_length": 1664.02001953125, |
| "completions/mean_terminated_length": 1466.212158203125, |
| "completions/min_length": 824.0, |
| "completions/min_terminated_length": 824.0, |
| "entropy": 0.27761187553405764, |
| "epoch": 0.0625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021949486806988716, |
| "learning_rate": 5e-05, |
| "loss": 0.11840050667524338, |
| "num_tokens": 2025232.0, |
| "reward": -0.2525097727775574, |
| "reward_std": 0.6308436393737793, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.8125097751617432, |
| "rewards/length_penalty/std": 0.18829484283924103, |
| "sampling/importance_sampling_ratio/max": 1.6358195543289185, |
| "sampling/importance_sampling_ratio/mean": 0.9903144836425781, |
| "sampling/importance_sampling_ratio/min": 0.5315372943878174, |
| "sampling/sampling_logp_difference/max": 0.6319819688796997, |
| "sampling/sampling_logp_difference/mean": 0.017899619415402412, |
| "step": 23, |
| "step_time": 28.560938307084143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005096927285194397, |
| "clip_ratio/high_mean": 0.0005096927285194397, |
| "clip_ratio/low_mean": 8.583837043261156e-05, |
| "clip_ratio/low_min": 8.583837043261156e-05, |
| "clip_ratio/region_mean": 0.0005955311004072428, |
| "completions/clipped_ratio": 0.4399999976158142, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2042.0, |
| "completions/mean_length": 1590.5599365234375, |
| "completions/mean_terminated_length": 1231.1429443359375, |
| "completions/min_length": 672.0, |
| "completions/min_terminated_length": 672.0, |
| "entropy": 0.23703448474407196, |
| "epoch": 0.06521739130434782, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019747884944081306, |
| "learning_rate": 5e-05, |
| "loss": 0.09910769760608673, |
| "num_tokens": 2107050.0, |
| "reward": -0.2166406214237213, |
| "reward_std": 0.7164656519889832, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.7766406536102295, |
| "rewards/length_penalty/std": 0.24764245748519897, |
| "sampling/importance_sampling_ratio/max": 1.457215428352356, |
| "sampling/importance_sampling_ratio/mean": 0.9916846752166748, |
| "sampling/importance_sampling_ratio/min": 0.5029337406158447, |
| "sampling/sampling_logp_difference/max": 0.6872968673706055, |
| "sampling/sampling_logp_difference/mean": 0.015448657795786858, |
| "step": 24, |
| "step_time": 27.499916886212304 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008482657605782152, |
| "clip_ratio/high_mean": 0.0008482657605782152, |
| "clip_ratio/low_mean": 0.00022012473782524468, |
| "clip_ratio/low_min": 0.00022012473782524468, |
| "clip_ratio/region_mean": 0.0010683904984034598, |
| "completions/clipped_ratio": 0.23999999463558197, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1919.0, |
| "completions/mean_length": 1442.97998046875, |
| "completions/mean_terminated_length": 1251.9210205078125, |
| "completions/min_length": 633.0, |
| "completions/min_terminated_length": 633.0, |
| "entropy": 0.2630555659532547, |
| "epoch": 0.06793478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021064141765236855, |
| "learning_rate": 5e-05, |
| "loss": 0.11863033473491669, |
| "num_tokens": 2181599.0, |
| "reward": 0.055419921875, |
| "reward_std": 0.6141738295555115, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.7045800685882568, |
| "rewards/length_penalty/std": 0.21538712084293365, |
| "sampling/importance_sampling_ratio/max": 1.5541377067565918, |
| "sampling/importance_sampling_ratio/mean": 0.990498960018158, |
| "sampling/importance_sampling_ratio/min": 0.40694311261177063, |
| "sampling/sampling_logp_difference/max": 0.8990819454193115, |
| "sampling/sampling_logp_difference/mean": 0.017718778923153877, |
| "step": 25, |
| "step_time": 26.41467908094637 |
| }, |
| { |
| "clip_ratio/high_max": 0.000819263607263565, |
| "clip_ratio/high_mean": 0.000819263607263565, |
| "clip_ratio/low_mean": 0.00011166772019350901, |
| "clip_ratio/low_min": 0.00011166772019350901, |
| "clip_ratio/region_mean": 0.0009309313609264791, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2045.0, |
| "completions/mean_length": 1262.239990234375, |
| "completions/mean_terminated_length": 1246.2041015625, |
| "completions/min_length": 687.0, |
| "completions/min_terminated_length": 687.0, |
| "entropy": 0.21845731139183044, |
| "epoch": 0.07065217391304347, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019048262387514114, |
| "learning_rate": 5e-05, |
| "loss": 0.11241159588098526, |
| "num_tokens": 2247181.0, |
| "reward": -0.056328125298023224, |
| "reward_std": 0.48785343766212463, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.6163281202316284, |
| "rewards/length_penalty/std": 0.16271910071372986, |
| "sampling/importance_sampling_ratio/max": 1.8472636938095093, |
| "sampling/importance_sampling_ratio/mean": 0.9921180605888367, |
| "sampling/importance_sampling_ratio/min": 0.45372647047042847, |
| "sampling/sampling_logp_difference/max": 0.7902607917785645, |
| "sampling/sampling_logp_difference/mean": 0.015436006709933281, |
| "step": 26, |
| "step_time": 26.193258133949712 |
| }, |
| { |
| "clip_ratio/high_max": 0.00045611696768901313, |
| "clip_ratio/high_mean": 0.00045611696768901313, |
| "clip_ratio/low_mean": 0.00010173475166084245, |
| "clip_ratio/low_min": 0.00010173475166084245, |
| "clip_ratio/region_mean": 0.0005578517040703446, |
| "completions/clipped_ratio": 0.3999999761581421, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2031.0, |
| "completions/mean_length": 1539.419921875, |
| "completions/mean_terminated_length": 1200.36669921875, |
| "completions/min_length": 202.0, |
| "completions/min_terminated_length": 202.0, |
| "entropy": 0.2073049932718277, |
| "epoch": 0.07336956521739131, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0178088266402483, |
| "learning_rate": 5e-05, |
| "loss": 0.05440214276313782, |
| "num_tokens": 2326402.0, |
| "reward": -0.1516699194908142, |
| "reward_std": 0.7180583477020264, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.7516699433326721, |
| "rewards/length_penalty/std": 0.26075154542922974, |
| "sampling/importance_sampling_ratio/max": 2.0076329708099365, |
| "sampling/importance_sampling_ratio/mean": 0.9929074645042419, |
| "sampling/importance_sampling_ratio/min": 0.3890666961669922, |
| "sampling/sampling_logp_difference/max": 0.9440045356750488, |
| "sampling/sampling_logp_difference/mean": 0.014081662520766258, |
| "step": 27, |
| "step_time": 26.73872535675764 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007280473771970719, |
| "clip_ratio/high_mean": 0.0007280473771970719, |
| "clip_ratio/low_mean": 6.700347003061325e-05, |
| "clip_ratio/low_min": 6.700347003061325e-05, |
| "clip_ratio/region_mean": 0.0007950508443173021, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1852.0, |
| "completions/mean_length": 1207.4599609375, |
| "completions/mean_terminated_length": 1190.30615234375, |
| "completions/min_length": 638.0, |
| "completions/min_terminated_length": 638.0, |
| "entropy": 0.1798260122537613, |
| "epoch": 0.07608695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016198398545384407, |
| "learning_rate": 5e-05, |
| "loss": 0.09480622410774231, |
| "num_tokens": 2390135.0, |
| "reward": 0.2104199230670929, |
| "reward_std": 0.4322068691253662, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.5895800590515137, |
| "rewards/length_penalty/std": 0.1593828648328781, |
| "sampling/importance_sampling_ratio/max": 1.6265084743499756, |
| "sampling/importance_sampling_ratio/mean": 0.993723452091217, |
| "sampling/importance_sampling_ratio/min": 0.38461166620254517, |
| "sampling/sampling_logp_difference/max": 0.9555211067199707, |
| "sampling/sampling_logp_difference/mean": 0.012959480285644531, |
| "step": 28, |
| "step_time": 25.803300652187318 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007129237637855113, |
| "clip_ratio/high_mean": 0.0007129237637855113, |
| "clip_ratio/low_mean": 0.00013153162435628474, |
| "clip_ratio/low_min": 0.00013153162435628474, |
| "clip_ratio/region_mean": 0.000844455393962562, |
| "completions/clipped_ratio": 0.3799999952316284, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2045.0, |
| "completions/mean_length": 1605.52001953125, |
| "completions/mean_terminated_length": 1334.322509765625, |
| "completions/min_length": 545.0, |
| "completions/min_terminated_length": 545.0, |
| "entropy": 0.23342860341072083, |
| "epoch": 0.07880434782608696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019112825393676758, |
| "learning_rate": 5e-05, |
| "loss": 0.08988890796899796, |
| "num_tokens": 2473851.0, |
| "reward": -0.08394531160593033, |
| "reward_std": 0.643122673034668, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.7839453220367432, |
| "rewards/length_penalty/std": 0.2589260935783386, |
| "sampling/importance_sampling_ratio/max": 2.976001739501953, |
| "sampling/importance_sampling_ratio/mean": 0.9917635917663574, |
| "sampling/importance_sampling_ratio/min": 0.3071954846382141, |
| "sampling/sampling_logp_difference/max": 1.180271029472351, |
| "sampling/sampling_logp_difference/mean": 0.016178693622350693, |
| "step": 29, |
| "step_time": 27.464959647972137 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007170933298766613, |
| "clip_ratio/high_mean": 0.0007170933298766613, |
| "clip_ratio/low_mean": 7.078182243276387e-05, |
| "clip_ratio/low_min": 7.078182243276387e-05, |
| "clip_ratio/region_mean": 0.0007878751552198082, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1948.0, |
| "completions/mean_length": 1215.739990234375, |
| "completions/mean_terminated_length": 1102.25, |
| "completions/min_length": 676.0, |
| "completions/min_terminated_length": 676.0, |
| "entropy": 0.1600349336862564, |
| "epoch": 0.08152173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01690404675900936, |
| "learning_rate": 5e-05, |
| "loss": 0.10917580127716064, |
| "num_tokens": 2536428.0, |
| "reward": 0.286376953125, |
| "reward_std": 0.49757134914398193, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826071977615356, |
| "rewards/length_penalty/mean": -0.5936230421066284, |
| "rewards/length_penalty/std": 0.20075169205665588, |
| "sampling/importance_sampling_ratio/max": 2.474348306655884, |
| "sampling/importance_sampling_ratio/mean": 0.9940484762191772, |
| "sampling/importance_sampling_ratio/min": 0.2940029501914978, |
| "sampling/sampling_logp_difference/max": 1.224165439605713, |
| "sampling/sampling_logp_difference/mean": 0.012239653617143631, |
| "step": 30, |
| "step_time": 25.22584995208308 |
| }, |
| { |
| "clip_ratio/high_max": 0.000856519874650985, |
| "clip_ratio/high_mean": 0.000856519874650985, |
| "clip_ratio/low_mean": 8.000027883099392e-05, |
| "clip_ratio/low_min": 8.000027883099392e-05, |
| "clip_ratio/region_mean": 0.0009365201694890857, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1861.0, |
| "completions/mean_length": 1463.8800048828125, |
| "completions/mean_terminated_length": 1317.8499755859375, |
| "completions/min_length": 510.0, |
| "completions/min_terminated_length": 510.0, |
| "entropy": 0.17065878212451935, |
| "epoch": 0.08423913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01730084978044033, |
| "learning_rate": 5e-05, |
| "loss": 0.12122836709022522, |
| "num_tokens": 2612752.0, |
| "reward": -0.29478514194488525, |
| "reward_std": 0.5411331653594971, |
| "rewards/correctness/mean": 0.41999998688697815, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.7147851586341858, |
| "rewards/length_penalty/std": 0.20178328454494476, |
| "sampling/importance_sampling_ratio/max": 2.579355478286743, |
| "sampling/importance_sampling_ratio/mean": 0.9938713908195496, |
| "sampling/importance_sampling_ratio/min": 0.2596469223499298, |
| "sampling/sampling_logp_difference/max": 1.3484325408935547, |
| "sampling/sampling_logp_difference/mean": 0.012738464400172234, |
| "step": 31, |
| "step_time": 27.176139670424163 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008451115863863379, |
| "clip_ratio/high_mean": 0.0008451115863863379, |
| "clip_ratio/low_mean": 0.00011503964924486354, |
| "clip_ratio/low_min": 0.00011503964924486354, |
| "clip_ratio/region_mean": 0.0009601512225344777, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2000.0, |
| "completions/mean_length": 1191.3399658203125, |
| "completions/mean_terminated_length": 1028.1666259765625, |
| "completions/min_length": 552.0, |
| "completions/min_terminated_length": 552.0, |
| "entropy": 0.21862303614616393, |
| "epoch": 0.08695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01855313777923584, |
| "learning_rate": 5e-05, |
| "loss": 0.1444239616394043, |
| "num_tokens": 2674249.0, |
| "reward": 0.23829101026058197, |
| "reward_std": 0.5837259888648987, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.5817089676856995, |
| "rewards/length_penalty/std": 0.2382737696170807, |
| "sampling/importance_sampling_ratio/max": 1.9388597011566162, |
| "sampling/importance_sampling_ratio/mean": 0.9922151565551758, |
| "sampling/importance_sampling_ratio/min": 0.24645483493804932, |
| "sampling/sampling_logp_difference/max": 1.4005765914916992, |
| "sampling/sampling_logp_difference/mean": 0.0162365585565567, |
| "step": 32, |
| "step_time": 25.234165051020682 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009740249835886062, |
| "clip_ratio/high_mean": 0.0009740249835886062, |
| "clip_ratio/low_mean": 5.943443829892203e-05, |
| "clip_ratio/low_min": 5.943443829892203e-05, |
| "clip_ratio/region_mean": 0.001033459440805018, |
| "completions/clipped_ratio": 0.25999999046325684, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1974.0, |
| "completions/mean_length": 1382.0599365234375, |
| "completions/mean_terminated_length": 1148.0810546875, |
| "completions/min_length": 706.0, |
| "completions/min_terminated_length": 706.0, |
| "entropy": 0.20403595566749572, |
| "epoch": 0.08967391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016193265095353127, |
| "learning_rate": 5e-05, |
| "loss": 0.05690198391675949, |
| "num_tokens": 2746992.0, |
| "reward": -0.09483398497104645, |
| "reward_std": 0.6409366726875305, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.6748340129852295, |
| "rewards/length_penalty/std": 0.2429748922586441, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9926742315292358, |
| "sampling/importance_sampling_ratio/min": 0.1798250824213028, |
| "sampling/sampling_logp_difference/max": 1.7157707214355469, |
| "sampling/sampling_logp_difference/mean": 0.014851015992462635, |
| "step": 33, |
| "step_time": 26.865770722040907 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012180538265965878, |
| "clip_ratio/high_mean": 0.0012180538265965878, |
| "clip_ratio/low_mean": 2.9444240499287844e-05, |
| "clip_ratio/low_min": 2.9444240499287844e-05, |
| "clip_ratio/region_mean": 0.0012474980554543435, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1734.0, |
| "completions/mean_length": 1317.179931640625, |
| "completions/mean_terminated_length": 1111.05126953125, |
| "completions/min_length": 647.0, |
| "completions/min_terminated_length": 647.0, |
| "entropy": 0.22754225134849548, |
| "epoch": 0.09239130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020097807049751282, |
| "learning_rate": 5e-05, |
| "loss": 0.0710531622171402, |
| "num_tokens": 2816741.0, |
| "reward": 0.11684569716453552, |
| "reward_std": 0.61951744556427, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.6431543231010437, |
| "rewards/length_penalty/std": 0.2254319041967392, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9915940761566162, |
| "sampling/importance_sampling_ratio/min": 0.23836448788642883, |
| "sampling/sampling_logp_difference/max": 1.4339543581008911, |
| "sampling/sampling_logp_difference/mean": 0.01691325753927231, |
| "step": 34, |
| "step_time": 26.204810510622337 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008569001860450953, |
| "clip_ratio/high_mean": 0.0008569001860450953, |
| "clip_ratio/low_mean": 0.00013024549843976274, |
| "clip_ratio/low_min": 0.00013024549843976274, |
| "clip_ratio/region_mean": 0.0009871456888504327, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2030.0, |
| "completions/mean_length": 1345.1400146484375, |
| "completions/mean_terminated_length": 1211.261962890625, |
| "completions/min_length": 632.0, |
| "completions/min_terminated_length": 632.0, |
| "entropy": 0.26689461171627044, |
| "epoch": 0.09510869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0222671777009964, |
| "learning_rate": 5e-05, |
| "loss": 0.15181683003902435, |
| "num_tokens": 2886728.0, |
| "reward": 0.1431933492422104, |
| "reward_std": 0.5873222351074219, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.6568066477775574, |
| "rewards/length_penalty/std": 0.22363705933094025, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.990323543548584, |
| "sampling/importance_sampling_ratio/min": 0.16977547109127045, |
| "sampling/sampling_logp_difference/max": 1.7732784748077393, |
| "sampling/sampling_logp_difference/mean": 0.01977042853832245, |
| "step": 35, |
| "step_time": 25.99194298312068 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007262625789735466, |
| "clip_ratio/high_mean": 0.0007262625789735466, |
| "clip_ratio/low_mean": 0.0001643200113903731, |
| "clip_ratio/low_min": 0.0001643200113903731, |
| "clip_ratio/region_mean": 0.0008905825903639198, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1975.0, |
| "completions/mean_length": 1117.199951171875, |
| "completions/mean_terminated_length": 990.2727661132812, |
| "completions/min_length": 532.0, |
| "completions/min_terminated_length": 532.0, |
| "entropy": 0.1896991550922394, |
| "epoch": 0.09782608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01747210882604122, |
| "learning_rate": 5e-05, |
| "loss": 0.058442991226911545, |
| "num_tokens": 2945058.0, |
| "reward": 0.33449217677116394, |
| "reward_std": 0.5202338695526123, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826071977615356, |
| "rewards/length_penalty/mean": -0.5455077886581421, |
| "rewards/length_penalty/std": 0.22712095081806183, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9930076599121094, |
| "sampling/importance_sampling_ratio/min": 0.1576518714427948, |
| "sampling/sampling_logp_difference/max": 1.8473659753799438, |
| "sampling/sampling_logp_difference/mean": 0.01649167202413082, |
| "step": 36, |
| "step_time": 25.64875900489278 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007515452452935278, |
| "clip_ratio/high_mean": 0.0007515452452935278, |
| "clip_ratio/low_mean": 0.00017000650841509924, |
| "clip_ratio/low_min": 0.00017000650841509924, |
| "clip_ratio/region_mean": 0.0009215517551638186, |
| "completions/clipped_ratio": 0.2199999988079071, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1888.0, |
| "completions/mean_length": 1225.93994140625, |
| "completions/mean_terminated_length": 994.0769653320312, |
| "completions/min_length": 605.0, |
| "completions/min_terminated_length": 605.0, |
| "entropy": 0.2357550859451294, |
| "epoch": 0.10054347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019464492797851562, |
| "learning_rate": 5e-05, |
| "loss": 0.09758821129798889, |
| "num_tokens": 3009875.0, |
| "reward": -0.01860351487994194, |
| "reward_std": 0.6543649435043335, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.5986034870147705, |
| "rewards/length_penalty/std": 0.24733325839042664, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9915767908096313, |
| "sampling/importance_sampling_ratio/min": 0.23241811990737915, |
| "sampling/sampling_logp_difference/max": 1.4991412162780762, |
| "sampling/sampling_logp_difference/mean": 0.01839047111570835, |
| "step": 37, |
| "step_time": 26.028307612985373 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007471224380424246, |
| "clip_ratio/high_mean": 0.0007471224380424246, |
| "clip_ratio/low_mean": 5.083655851194635e-05, |
| "clip_ratio/low_min": 5.083655851194635e-05, |
| "clip_ratio/region_mean": 0.0007979590183822439, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1990.0, |
| "completions/mean_length": 1287.699951171875, |
| "completions/mean_terminated_length": 1203.2222900390625, |
| "completions/min_length": 667.0, |
| "completions/min_terminated_length": 667.0, |
| "entropy": 0.1729876458644867, |
| "epoch": 0.10326086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01705159805715084, |
| "learning_rate": 5e-05, |
| "loss": 0.07221940904855728, |
| "num_tokens": 3077650.0, |
| "reward": 0.25124022364616394, |
| "reward_std": 0.4530826508998871, |
| "rewards/correctness/mean": 0.8799999952316284, |
| "rewards/correctness/std": 0.32826074957847595, |
| "rewards/length_penalty/mean": -0.6287597417831421, |
| "rewards/length_penalty/std": 0.2424236387014389, |
| "sampling/importance_sampling_ratio/max": 2.784482002258301, |
| "sampling/importance_sampling_ratio/mean": 0.9937966465950012, |
| "sampling/importance_sampling_ratio/min": 0.0835733637213707, |
| "sampling/sampling_logp_difference/max": 2.4820303916931152, |
| "sampling/sampling_logp_difference/mean": 0.014295706525444984, |
| "step": 38, |
| "step_time": 26.3849402081687 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047908292326610536, |
| "clip_ratio/high_mean": 0.00047908292326610536, |
| "clip_ratio/low_mean": 6.898379942867904e-05, |
| "clip_ratio/low_min": 6.898379942867904e-05, |
| "clip_ratio/region_mean": 0.0005480667285155505, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1733.0, |
| "completions/mean_length": 1131.1400146484375, |
| "completions/mean_terminated_length": 929.8779907226562, |
| "completions/min_length": 473.0, |
| "completions/min_terminated_length": 473.0, |
| "entropy": 0.18207353055477143, |
| "epoch": 0.10597826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02022084966301918, |
| "learning_rate": 5e-05, |
| "loss": 0.05686383694410324, |
| "num_tokens": 3136697.0, |
| "reward": 0.06768554449081421, |
| "reward_std": 0.7044277191162109, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.5523144602775574, |
| "rewards/length_penalty/std": 0.25435444712638855, |
| "sampling/importance_sampling_ratio/max": 2.187497615814209, |
| "sampling/importance_sampling_ratio/mean": 0.993350088596344, |
| "sampling/importance_sampling_ratio/min": 0.27275171875953674, |
| "sampling/sampling_logp_difference/max": 1.2991933822631836, |
| "sampling/sampling_logp_difference/mean": 0.014788741245865822, |
| "step": 39, |
| "step_time": 24.97533990885131 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007706037955358625, |
| "clip_ratio/high_mean": 0.0007706037955358625, |
| "clip_ratio/low_mean": 9.303098486270755e-05, |
| "clip_ratio/low_min": 9.303098486270755e-05, |
| "clip_ratio/region_mean": 0.0008636347949504853, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1812.0, |
| "completions/max_terminated_length": 1812.0, |
| "completions/mean_length": 870.8999633789062, |
| "completions/mean_terminated_length": 870.8999633789062, |
| "completions/min_length": 654.0, |
| "completions/min_terminated_length": 654.0, |
| "entropy": 0.1895820140838623, |
| "epoch": 0.10869565217391304, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014811172150075436, |
| "learning_rate": 5e-05, |
| "loss": 0.06211411952972412, |
| "num_tokens": 3182962.0, |
| "reward": 0.574755847454071, |
| "reward_std": 0.1076437383890152, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.42524415254592896, |
| "rewards/length_penalty/std": 0.1076437383890152, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.993091881275177, |
| "sampling/importance_sampling_ratio/min": 0.11348514258861542, |
| "sampling/sampling_logp_difference/max": 2.1760833263397217, |
| "sampling/sampling_logp_difference/mean": 0.017822640016674995, |
| "step": 40, |
| "step_time": 21.239220429910347 |
| }, |
| { |
| "clip_ratio/high_max": 0.000729369098553434, |
| "clip_ratio/high_mean": 0.000729369098553434, |
| "clip_ratio/low_mean": 6.442431767936796e-05, |
| "clip_ratio/low_min": 6.442431767936796e-05, |
| "clip_ratio/region_mean": 0.0007937934075016529, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1935.0, |
| "completions/mean_length": 1214.1199951171875, |
| "completions/mean_terminated_length": 1100.4091796875, |
| "completions/min_length": 598.0, |
| "completions/min_terminated_length": 598.0, |
| "entropy": 0.20386375486850739, |
| "epoch": 0.11141304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0198789294809103, |
| "learning_rate": 5e-05, |
| "loss": 0.09383414685726166, |
| "num_tokens": 3247408.0, |
| "reward": -0.05283202975988388, |
| "reward_std": 0.612436830997467, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.5928320288658142, |
| "rewards/length_penalty/std": 0.20813514292240143, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9923471212387085, |
| "sampling/importance_sampling_ratio/min": 0.11420464515686035, |
| "sampling/sampling_logp_difference/max": 2.1697633266448975, |
| "sampling/sampling_logp_difference/mean": 0.017043285071849823, |
| "step": 41, |
| "step_time": 25.912487969035283 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005029736028518527, |
| "clip_ratio/high_mean": 0.0005029736028518527, |
| "clip_ratio/low_mean": 0.00012810062034986913, |
| "clip_ratio/low_min": 0.00012810062034986913, |
| "clip_ratio/region_mean": 0.0006310742348432541, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1992.0, |
| "completions/mean_length": 1178.3599853515625, |
| "completions/mean_terminated_length": 960.9500122070312, |
| "completions/min_length": 471.0, |
| "completions/min_terminated_length": 471.0, |
| "entropy": 0.20278692245483398, |
| "epoch": 0.11413043478260869, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01870260015130043, |
| "learning_rate": 5e-05, |
| "loss": 0.03833295777440071, |
| "num_tokens": 3309316.0, |
| "reward": 0.22462889552116394, |
| "reward_std": 0.6349654197692871, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5753710865974426, |
| "rewards/length_penalty/std": 0.2908514440059662, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9925538301467896, |
| "sampling/importance_sampling_ratio/min": 0.16318759322166443, |
| "sampling/sampling_logp_difference/max": 1.8128548860549927, |
| "sampling/sampling_logp_difference/mean": 0.01731746457517147, |
| "step": 42, |
| "step_time": 26.0047237186227 |
| }, |
| { |
| "clip_ratio/high_max": 0.001067713089287281, |
| "clip_ratio/high_mean": 0.001067713089287281, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001067713089287281, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1500.0, |
| "completions/max_terminated_length": 1500.0, |
| "completions/mean_length": 777.4400024414062, |
| "completions/mean_terminated_length": 777.4400024414062, |
| "completions/min_length": 481.0, |
| "completions/min_terminated_length": 481.0, |
| "entropy": 0.2059490293264389, |
| "epoch": 0.11684782608695653, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01624198630452156, |
| "learning_rate": 5e-05, |
| "loss": 0.05506262555718422, |
| "num_tokens": 3350158.0, |
| "reward": 0.600390613079071, |
| "reward_std": 0.18859446048736572, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.3796093761920929, |
| "rewards/length_penalty/std": 0.12124241143465042, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9921716451644897, |
| "sampling/importance_sampling_ratio/min": 0.048016149550676346, |
| "sampling/sampling_logp_difference/max": 3.0362179279327393, |
| "sampling/sampling_logp_difference/mean": 0.021691124886274338, |
| "step": 43, |
| "step_time": 17.796076989499852 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006964510248508304, |
| "clip_ratio/high_mean": 0.0006964510248508304, |
| "clip_ratio/low_mean": 0.00017296440637437627, |
| "clip_ratio/low_min": 0.00017296440637437627, |
| "clip_ratio/region_mean": 0.0008694154501426965, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1803.0, |
| "completions/mean_length": 1044.780029296875, |
| "completions/mean_terminated_length": 1024.30615234375, |
| "completions/min_length": 441.0, |
| "completions/min_terminated_length": 441.0, |
| "entropy": 0.1765652060508728, |
| "epoch": 0.11956521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015211697667837143, |
| "learning_rate": 5e-05, |
| "loss": 0.09405821561813354, |
| "num_tokens": 3405397.0, |
| "reward": 0.4698534905910492, |
| "reward_std": 0.28211453557014465, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.5101464986801147, |
| "rewards/length_penalty/std": 0.19898390769958496, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9933960437774658, |
| "sampling/importance_sampling_ratio/min": 0.05977986007928848, |
| "sampling/sampling_logp_difference/max": 2.8170864582061768, |
| "sampling/sampling_logp_difference/mean": 0.016818420961499214, |
| "step": 44, |
| "step_time": 24.691332526504993 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007825112727005035, |
| "clip_ratio/high_mean": 0.0007825112727005035, |
| "clip_ratio/low_mean": 3.903874894604087e-05, |
| "clip_ratio/low_min": 3.903874894604087e-05, |
| "clip_ratio/region_mean": 0.0008215500216465444, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1795.0, |
| "completions/mean_length": 1010.0199584960938, |
| "completions/mean_terminated_length": 966.7708740234375, |
| "completions/min_length": 430.0, |
| "completions/min_terminated_length": 430.0, |
| "entropy": 0.2038260132074356, |
| "epoch": 0.12228260869565218, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016963917762041092, |
| "learning_rate": 5e-05, |
| "loss": 0.05373412370681763, |
| "num_tokens": 3458838.0, |
| "reward": 0.28682616353034973, |
| "reward_std": 0.5763336420059204, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.49317383766174316, |
| "rewards/length_penalty/std": 0.2312302142381668, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9924213886260986, |
| "sampling/importance_sampling_ratio/min": 0.06157507002353668, |
| "sampling/sampling_logp_difference/max": 2.7874982357025146, |
| "sampling/sampling_logp_difference/mean": 0.018136093392968178, |
| "step": 45, |
| "step_time": 25.164567972067744 |
| }, |
| { |
| "clip_ratio/high_max": 0.00039000823890091854, |
| "clip_ratio/high_mean": 0.00039000823890091854, |
| "clip_ratio/low_mean": 7.269879861269147e-05, |
| "clip_ratio/low_min": 7.269879861269147e-05, |
| "clip_ratio/region_mean": 0.0004627070476999506, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1188.0, |
| "completions/mean_length": 1086.6400146484375, |
| "completions/mean_terminated_length": 846.2999877929688, |
| "completions/min_length": 463.0, |
| "completions/min_terminated_length": 463.0, |
| "entropy": 0.22776823937892915, |
| "epoch": 0.125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015369636937975883, |
| "learning_rate": 5e-05, |
| "loss": 0.04503121227025986, |
| "num_tokens": 3516680.0, |
| "reward": 0.2694140672683716, |
| "reward_std": 0.6474911570549011, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5305859446525574, |
| "rewards/length_penalty/std": 0.25373587012290955, |
| "sampling/importance_sampling_ratio/max": 2.9789910316467285, |
| "sampling/importance_sampling_ratio/mean": 0.9916025996208191, |
| "sampling/importance_sampling_ratio/min": 0.04188171774148941, |
| "sampling/sampling_logp_difference/max": 3.172905921936035, |
| "sampling/sampling_logp_difference/mean": 0.019508102908730507, |
| "step": 46, |
| "step_time": 25.130053090164438 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006507517013233155, |
| "clip_ratio/high_mean": 0.0006507517013233155, |
| "clip_ratio/low_mean": 0.00011765561066567898, |
| "clip_ratio/low_min": 0.00011765561066567898, |
| "clip_ratio/region_mean": 0.0007684073178097605, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1485.0, |
| "completions/mean_length": 826.6199951171875, |
| "completions/mean_terminated_length": 775.7291870117188, |
| "completions/min_length": 500.0, |
| "completions/min_terminated_length": 500.0, |
| "entropy": 0.175468310713768, |
| "epoch": 0.12771739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01958637870848179, |
| "learning_rate": 5e-05, |
| "loss": 0.06920386850833893, |
| "num_tokens": 3560401.0, |
| "reward": 0.5163769125938416, |
| "reward_std": 0.41254669427871704, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.4036230444908142, |
| "rewards/length_penalty/std": 0.16869166493415833, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.993443489074707, |
| "sampling/importance_sampling_ratio/min": 0.13633184134960175, |
| "sampling/sampling_logp_difference/max": 1.9926633834838867, |
| "sampling/sampling_logp_difference/mean": 0.01789533719420433, |
| "step": 47, |
| "step_time": 23.82672528317198 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005032541987020523, |
| "clip_ratio/high_mean": 0.0005032541987020523, |
| "clip_ratio/low_mean": 6.262429669732228e-05, |
| "clip_ratio/low_min": 6.262429669732228e-05, |
| "clip_ratio/region_mean": 0.0005658784910337999, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1960.0, |
| "completions/mean_length": 1017.1199951171875, |
| "completions/mean_terminated_length": 927.478271484375, |
| "completions/min_length": 458.0, |
| "completions/min_terminated_length": 458.0, |
| "entropy": 0.1553356945514679, |
| "epoch": 0.13043478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017065836116671562, |
| "learning_rate": 5e-05, |
| "loss": 0.05763271450996399, |
| "num_tokens": 3614217.0, |
| "reward": 0.18335936963558197, |
| "reward_std": 0.617246687412262, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.47121208906173706, |
| "rewards/length_penalty/mean": -0.4966406226158142, |
| "rewards/length_penalty/std": 0.2483815848827362, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9945675134658813, |
| "sampling/importance_sampling_ratio/min": 0.1440926343202591, |
| "sampling/sampling_logp_difference/max": 2.313066005706787, |
| "sampling/sampling_logp_difference/mean": 0.014977255836129189, |
| "step": 48, |
| "step_time": 25.213895262451842 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016091061057522892, |
| "clip_ratio/high_mean": 0.0016091061057522892, |
| "clip_ratio/low_mean": 4.549590521492064e-05, |
| "clip_ratio/low_min": 4.549590521492064e-05, |
| "clip_ratio/region_mean": 0.0016546020051464438, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1164.0, |
| "completions/mean_length": 972.0799560546875, |
| "completions/mean_terminated_length": 703.1000366210938, |
| "completions/min_length": 389.0, |
| "completions/min_terminated_length": 389.0, |
| "entropy": 0.24074310958385467, |
| "epoch": 0.1331521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01453313510864973, |
| "learning_rate": 5e-05, |
| "loss": 0.04727436602115631, |
| "num_tokens": 3665511.0, |
| "reward": 0.3253515660762787, |
| "reward_std": 0.6757630109786987, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.47464844584465027, |
| "rewards/length_penalty/std": 0.2810002267360687, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.991299033164978, |
| "sampling/importance_sampling_ratio/min": 0.12447156012058258, |
| "sampling/sampling_logp_difference/max": 2.0836780071258545, |
| "sampling/sampling_logp_difference/mean": 0.021517034620046616, |
| "step": 49, |
| "step_time": 24.520182807929814 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003247870015911758, |
| "clip_ratio/high_mean": 0.0003247870015911758, |
| "clip_ratio/low_mean": 0.00028453292179619893, |
| "clip_ratio/low_min": 0.00028453292179619893, |
| "clip_ratio/region_mean": 0.0006093199306633323, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1019.0, |
| "completions/mean_length": 912.2799682617188, |
| "completions/mean_terminated_length": 628.3500366210938, |
| "completions/min_length": 287.0, |
| "completions/min_terminated_length": 287.0, |
| "entropy": 0.25229776501655576, |
| "epoch": 0.1358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013235251419246197, |
| "learning_rate": 5e-05, |
| "loss": 0.02668197639286518, |
| "num_tokens": 3713535.0, |
| "reward": 0.15455077588558197, |
| "reward_std": 0.7314342856407166, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.44544923305511475, |
| "rewards/length_penalty/std": 0.29221686720848083, |
| "sampling/importance_sampling_ratio/max": 2.721278667449951, |
| "sampling/importance_sampling_ratio/mean": 0.9899016618728638, |
| "sampling/importance_sampling_ratio/min": 0.042893968522548676, |
| "sampling/sampling_logp_difference/max": 3.14902400970459, |
| "sampling/sampling_logp_difference/mean": 0.022883890196681023, |
| "step": 50, |
| "step_time": 24.369463649811223 |
| }, |
| { |
| "clip_ratio/high_max": 0.000643822574056685, |
| "clip_ratio/high_mean": 0.000643822574056685, |
| "clip_ratio/low_mean": 6.33914431091398e-05, |
| "clip_ratio/low_min": 6.33914431091398e-05, |
| "clip_ratio/region_mean": 0.0007072140229865909, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1970.0, |
| "completions/max_terminated_length": 1970.0, |
| "completions/mean_length": 766.9599609375, |
| "completions/mean_terminated_length": 766.9599609375, |
| "completions/min_length": 382.0, |
| "completions/min_terminated_length": 382.0, |
| "entropy": 0.16464298963546753, |
| "epoch": 0.13858695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015319639816880226, |
| "learning_rate": 5e-05, |
| "loss": 0.06438955664634705, |
| "num_tokens": 3754393.0, |
| "reward": 0.6255077719688416, |
| "reward_std": 0.1482277661561966, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.37449219822883606, |
| "rewards/length_penalty/std": 0.1482277661561966, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9936192035675049, |
| "sampling/importance_sampling_ratio/min": 0.11222824454307556, |
| "sampling/sampling_logp_difference/max": 2.187220573425293, |
| "sampling/sampling_logp_difference/mean": 0.018299877643585205, |
| "step": 51, |
| "step_time": 22.948548633838072 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007740166562143713, |
| "clip_ratio/high_mean": 0.0007740166562143713, |
| "clip_ratio/low_mean": 0.0001623515330720693, |
| "clip_ratio/low_min": 0.0001623515330720693, |
| "clip_ratio/region_mean": 0.0009363682067487389, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1427.0, |
| "completions/max_terminated_length": 1427.0, |
| "completions/mean_length": 700.0799560546875, |
| "completions/mean_terminated_length": 700.0799560546875, |
| "completions/min_length": 454.0, |
| "completions/min_terminated_length": 454.0, |
| "entropy": 0.15744704604148865, |
| "epoch": 0.14130434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012992658652365208, |
| "learning_rate": 5e-05, |
| "loss": 0.04073449596762657, |
| "num_tokens": 3791657.0, |
| "reward": 0.4581640660762787, |
| "reward_std": 0.41504931449890137, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.34183594584465027, |
| "rewards/length_penalty/std": 0.11200068145990372, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9942061305046082, |
| "sampling/importance_sampling_ratio/min": 0.15024611353874207, |
| "sampling/sampling_logp_difference/max": 1.8954805135726929, |
| "sampling/sampling_logp_difference/mean": 0.019554542377591133, |
| "step": 52, |
| "step_time": 16.728716250276193 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006250745645957068, |
| "clip_ratio/high_mean": 0.0006250745645957068, |
| "clip_ratio/low_mean": 2.658160519786179e-05, |
| "clip_ratio/low_min": 2.658160519786179e-05, |
| "clip_ratio/region_mean": 0.0006516561697935686, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1354.0, |
| "completions/mean_length": 807.3599853515625, |
| "completions/mean_terminated_length": 755.6666870117188, |
| "completions/min_length": 397.0, |
| "completions/min_terminated_length": 397.0, |
| "entropy": 0.1829791933298111, |
| "epoch": 0.14402173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021517958492040634, |
| "learning_rate": 5e-05, |
| "loss": 0.10340322554111481, |
| "num_tokens": 3834745.0, |
| "reward": 0.3657812476158142, |
| "reward_std": 0.571151077747345, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.3942187428474426, |
| "rewards/length_penalty/std": 0.1765245497226715, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9934973120689392, |
| "sampling/importance_sampling_ratio/min": 0.05403505638241768, |
| "sampling/sampling_logp_difference/max": 2.9181222915649414, |
| "sampling/sampling_logp_difference/mean": 0.021358292549848557, |
| "step": 53, |
| "step_time": 23.810034946305677 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007082508644089102, |
| "clip_ratio/high_mean": 0.0007082508644089102, |
| "clip_ratio/low_mean": 0.00010528063430683687, |
| "clip_ratio/low_min": 0.00010528063430683687, |
| "clip_ratio/region_mean": 0.0008135314856190234, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1746.0, |
| "completions/mean_length": 1111.3199462890625, |
| "completions/mean_terminated_length": 983.5909423828125, |
| "completions/min_length": 512.0, |
| "completions/min_terminated_length": 512.0, |
| "entropy": 0.1832393229007721, |
| "epoch": 0.14673913043478262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019397616386413574, |
| "learning_rate": 5e-05, |
| "loss": 0.11991956830024719, |
| "num_tokens": 3894131.0, |
| "reward": 0.31736326217651367, |
| "reward_std": 0.5375475883483887, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.5426366925239563, |
| "rewards/length_penalty/std": 0.23527614772319794, |
| "sampling/importance_sampling_ratio/max": 2.2643704414367676, |
| "sampling/importance_sampling_ratio/mean": 0.9930412769317627, |
| "sampling/importance_sampling_ratio/min": 0.04445350170135498, |
| "sampling/sampling_logp_difference/max": 3.113311529159546, |
| "sampling/sampling_logp_difference/mean": 0.01802951470017433, |
| "step": 54, |
| "step_time": 25.670479513239115 |
| }, |
| { |
| "clip_ratio/high_max": 0.000749661133158952, |
| "clip_ratio/high_mean": 0.000749661133158952, |
| "clip_ratio/low_mean": 0.00012117865699110553, |
| "clip_ratio/low_min": 0.00012117865699110553, |
| "clip_ratio/region_mean": 0.0008708397857844829, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2039.0, |
| "completions/mean_length": 1065.1199951171875, |
| "completions/mean_terminated_length": 931.0909423828125, |
| "completions/min_length": 480.0, |
| "completions/min_terminated_length": 480.0, |
| "entropy": 0.22501038908958435, |
| "epoch": 0.14945652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021781066432595253, |
| "learning_rate": 5e-05, |
| "loss": 0.07857176661491394, |
| "num_tokens": 3950277.0, |
| "reward": 0.3199218809604645, |
| "reward_std": 0.5881868600845337, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.5200781226158142, |
| "rewards/length_penalty/std": 0.28003421425819397, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9916319251060486, |
| "sampling/importance_sampling_ratio/min": 0.04320801421999931, |
| "sampling/sampling_logp_difference/max": 3.1417293548583984, |
| "sampling/sampling_logp_difference/mean": 0.021369094029068947, |
| "step": 55, |
| "step_time": 24.98296606983058 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002201356299337931, |
| "clip_ratio/high_mean": 0.0002201356299337931, |
| "clip_ratio/low_mean": 0.0001502036669990048, |
| "clip_ratio/low_min": 0.0001502036669990048, |
| "clip_ratio/region_mean": 0.000370339305663947, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1660.0, |
| "completions/mean_length": 928.5, |
| "completions/mean_terminated_length": 648.625, |
| "completions/min_length": 346.0, |
| "completions/min_terminated_length": 346.0, |
| "entropy": 0.2072504162788391, |
| "epoch": 0.15217391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013964945450425148, |
| "learning_rate": 5e-05, |
| "loss": 0.05491746589541435, |
| "num_tokens": 3999772.0, |
| "reward": 0.14663085341453552, |
| "reward_std": 0.6885973215103149, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.453369140625, |
| "rewards/length_penalty/std": 0.30666640400886536, |
| "sampling/importance_sampling_ratio/max": 2.906921625137329, |
| "sampling/importance_sampling_ratio/mean": 0.9920700788497925, |
| "sampling/importance_sampling_ratio/min": 0.03219592198729515, |
| "sampling/sampling_logp_difference/max": 3.435915470123291, |
| "sampling/sampling_logp_difference/mean": 0.02083013206720352, |
| "step": 56, |
| "step_time": 24.479757665889338 |
| }, |
| { |
| "clip_ratio/high_max": 0.001786479854490608, |
| "clip_ratio/high_mean": 0.001786479854490608, |
| "clip_ratio/low_mean": 2.5886617368087173e-05, |
| "clip_ratio/low_min": 2.5886617368087173e-05, |
| "clip_ratio/region_mean": 0.0018123664776794612, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1872.0, |
| "completions/mean_length": 1050.699951171875, |
| "completions/mean_terminated_length": 801.375, |
| "completions/min_length": 299.0, |
| "completions/min_terminated_length": 299.0, |
| "entropy": 0.2278908148407936, |
| "epoch": 0.15489130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014048721641302109, |
| "learning_rate": 5e-05, |
| "loss": 0.037405215203762054, |
| "num_tokens": 4055547.0, |
| "reward": 0.2869628965854645, |
| "reward_std": 0.6820490956306458, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.5130370855331421, |
| "rewards/length_penalty/std": 0.32119351625442505, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.991212785243988, |
| "sampling/importance_sampling_ratio/min": 0.018531763926148415, |
| "sampling/sampling_logp_difference/max": 3.988269090652466, |
| "sampling/sampling_logp_difference/mean": 0.021337997168302536, |
| "step": 57, |
| "step_time": 24.977965974714607 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005700029694708064, |
| "clip_ratio/high_mean": 0.0005700029694708064, |
| "clip_ratio/low_mean": 0.00015229834825731815, |
| "clip_ratio/low_min": 0.00015229834825731815, |
| "clip_ratio/region_mean": 0.0007223013206385076, |
| "completions/clipped_ratio": 0.25999999046325684, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1868.0, |
| "completions/mean_length": 1198.6199951171875, |
| "completions/mean_terminated_length": 900.189208984375, |
| "completions/min_length": 466.0, |
| "completions/min_terminated_length": 466.0, |
| "entropy": 0.255254665017128, |
| "epoch": 0.15760869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01905212551355362, |
| "learning_rate": 5e-05, |
| "loss": 0.09029528498649597, |
| "num_tokens": 4119288.0, |
| "reward": -0.0052636717446148396, |
| "reward_std": 0.7467082142829895, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.5852636694908142, |
| "rewards/length_penalty/std": 0.3044397234916687, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9905498623847961, |
| "sampling/importance_sampling_ratio/min": 0.05557810142636299, |
| "sampling/sampling_logp_difference/max": 2.8899660110473633, |
| "sampling/sampling_logp_difference/mean": 0.023080680519342422, |
| "step": 58, |
| "step_time": 26.597216782160103 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008118896570522338, |
| "clip_ratio/high_mean": 0.0008118896570522338, |
| "clip_ratio/low_mean": 0.00015391591878142208, |
| "clip_ratio/low_min": 0.00015391591878142208, |
| "clip_ratio/region_mean": 0.0009658055671025068, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1400.0, |
| "completions/mean_length": 677.4199829101562, |
| "completions/mean_terminated_length": 649.448974609375, |
| "completions/min_length": 304.0, |
| "completions/min_terminated_length": 304.0, |
| "entropy": 0.14910456240177156, |
| "epoch": 0.16032608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016270486637949944, |
| "learning_rate": 5e-05, |
| "loss": 0.04443974792957306, |
| "num_tokens": 4155869.0, |
| "reward": 0.5692285299301147, |
| "reward_std": 0.390527606010437, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304574966430664, |
| "rewards/length_penalty/mean": -0.33077147603034973, |
| "rewards/length_penalty/std": 0.146135151386261, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9942877888679504, |
| "sampling/importance_sampling_ratio/min": 0.08243148028850555, |
| "sampling/sampling_logp_difference/max": 2.4957878589630127, |
| "sampling/sampling_logp_difference/mean": 0.019903726875782013, |
| "step": 59, |
| "step_time": 22.8233163098339 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006108953821239993, |
| "clip_ratio/high_mean": 0.0006108953821239993, |
| "clip_ratio/low_mean": 8.367137343157083e-05, |
| "clip_ratio/low_min": 8.367137343157083e-05, |
| "clip_ratio/region_mean": 0.0006945667759282514, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1135.0, |
| "completions/max_terminated_length": 1135.0, |
| "completions/mean_length": 740.719970703125, |
| "completions/mean_terminated_length": 740.719970703125, |
| "completions/min_length": 340.0, |
| "completions/min_terminated_length": 340.0, |
| "entropy": 0.1358457773923874, |
| "epoch": 0.16304347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01377896312624216, |
| "learning_rate": 5e-05, |
| "loss": 0.04055830091238022, |
| "num_tokens": 4195715.0, |
| "reward": 0.018320312723517418, |
| "reward_std": 0.4930758476257324, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.36167967319488525, |
| "rewards/length_penalty/std": 0.10894857347011566, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9946678876876831, |
| "sampling/importance_sampling_ratio/min": 0.06510572135448456, |
| "sampling/sampling_logp_difference/max": 2.7317428588867188, |
| "sampling/sampling_logp_difference/mean": 0.01834091730415821, |
| "step": 60, |
| "step_time": 14.378700571134686 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007366885984083638, |
| "clip_ratio/high_mean": 0.0007366885984083638, |
| "clip_ratio/low_mean": 0.00010070936405099928, |
| "clip_ratio/low_min": 0.00010070936405099928, |
| "clip_ratio/region_mean": 0.0008373979799216613, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 896.0, |
| "completions/max_terminated_length": 896.0, |
| "completions/mean_length": 608.1199951171875, |
| "completions/mean_terminated_length": 608.1199951171875, |
| "completions/min_length": 399.0, |
| "completions/min_terminated_length": 399.0, |
| "entropy": 0.1377216622233391, |
| "epoch": 0.16576086956521738, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013815459795296192, |
| "learning_rate": 5e-05, |
| "loss": 0.033973757177591324, |
| "num_tokens": 4228531.0, |
| "reward": 0.7030664086341858, |
| "reward_std": 0.04950352758169174, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.2969335913658142, |
| "rewards/length_penalty/std": 0.049503523856401443, |
| "sampling/importance_sampling_ratio/max": 2.5220999717712402, |
| "sampling/importance_sampling_ratio/mean": 0.9949359893798828, |
| "sampling/importance_sampling_ratio/min": 0.05055711790919304, |
| "sampling/sampling_logp_difference/max": 2.984651565551758, |
| "sampling/sampling_logp_difference/mean": 0.019737115129828453, |
| "step": 61, |
| "step_time": 11.692759637953714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006575177831109613, |
| "clip_ratio/high_mean": 0.0006575177831109613, |
| "clip_ratio/low_mean": 6.264255061978474e-05, |
| "clip_ratio/low_min": 6.264255061978474e-05, |
| "clip_ratio/region_mean": 0.0007201603322755546, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1995.0, |
| "completions/max_terminated_length": 1995.0, |
| "completions/mean_length": 890.6199951171875, |
| "completions/mean_terminated_length": 890.6199951171875, |
| "completions/min_length": 326.0, |
| "completions/min_terminated_length": 326.0, |
| "entropy": 0.14189326465129853, |
| "epoch": 0.16847826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015275136567652225, |
| "learning_rate": 5e-05, |
| "loss": 0.03717654570937157, |
| "num_tokens": 4276682.0, |
| "reward": 0.4051269292831421, |
| "reward_std": 0.45795977115631104, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.4348730444908142, |
| "rewards/length_penalty/std": 0.19076396524906158, |
| "sampling/importance_sampling_ratio/max": 2.618985414505005, |
| "sampling/importance_sampling_ratio/mean": 0.9940707683563232, |
| "sampling/importance_sampling_ratio/min": 0.038013607263565063, |
| "sampling/sampling_logp_difference/max": 3.2698111534118652, |
| "sampling/sampling_logp_difference/mean": 0.016751455143094063, |
| "step": 62, |
| "step_time": 23.50743620051071 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008010541961994023, |
| "clip_ratio/high_mean": 0.0008010541961994023, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008010541961994023, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1916.0, |
| "completions/mean_length": 706.97998046875, |
| "completions/mean_terminated_length": 679.6122436523438, |
| "completions/min_length": 219.0, |
| "completions/min_terminated_length": 219.0, |
| "entropy": 0.1526895821094513, |
| "epoch": 0.17119565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01731613650918007, |
| "learning_rate": 5e-05, |
| "loss": 0.04923119395971298, |
| "num_tokens": 4315281.0, |
| "reward": 0.45479491353034973, |
| "reward_std": 0.5850844383239746, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.34520506858825684, |
| "rewards/length_penalty/std": 0.20253480970859528, |
| "sampling/importance_sampling_ratio/max": 2.8465676307678223, |
| "sampling/importance_sampling_ratio/mean": 0.9949560165405273, |
| "sampling/importance_sampling_ratio/min": 0.27247118949890137, |
| "sampling/sampling_logp_difference/max": 1.300222396850586, |
| "sampling/sampling_logp_difference/mean": 0.019272862002253532, |
| "step": 63, |
| "step_time": 23.484382715076208 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003500981023535132, |
| "clip_ratio/high_mean": 0.0003500981023535132, |
| "clip_ratio/low_mean": 5.712283600587398e-05, |
| "clip_ratio/low_min": 5.712283600587398e-05, |
| "clip_ratio/region_mean": 0.00040722094126977026, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1117.0, |
| "completions/max_terminated_length": 1117.0, |
| "completions/mean_length": 700.2999877929688, |
| "completions/mean_terminated_length": 700.2999877929688, |
| "completions/min_length": 324.0, |
| "completions/min_terminated_length": 324.0, |
| "entropy": 0.13254741728305816, |
| "epoch": 0.17391304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01813482865691185, |
| "learning_rate": 5e-05, |
| "loss": 0.023750657215714455, |
| "num_tokens": 4353816.0, |
| "reward": 0.6180566549301147, |
| "reward_std": 0.22116827964782715, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.3419433534145355, |
| "rewards/length_penalty/std": 0.09352646768093109, |
| "sampling/importance_sampling_ratio/max": 2.9889914989471436, |
| "sampling/importance_sampling_ratio/mean": 0.9951876401901245, |
| "sampling/importance_sampling_ratio/min": 0.07279559969902039, |
| "sampling/sampling_logp_difference/max": 2.6200997829437256, |
| "sampling/sampling_logp_difference/mean": 0.018573181703686714, |
| "step": 64, |
| "step_time": 13.950854809721932 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010051134566310793, |
| "clip_ratio/high_mean": 0.0010051134566310793, |
| "clip_ratio/low_mean": 7.008167740423233e-05, |
| "clip_ratio/low_min": 7.008167740423233e-05, |
| "clip_ratio/region_mean": 0.0010751951369456947, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1941.0, |
| "completions/max_terminated_length": 1941.0, |
| "completions/mean_length": 760.8599853515625, |
| "completions/mean_terminated_length": 760.8599853515625, |
| "completions/min_length": 306.0, |
| "completions/min_terminated_length": 306.0, |
| "entropy": 0.15310844779014587, |
| "epoch": 0.1766304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01888042315840721, |
| "learning_rate": 5e-05, |
| "loss": 0.04399148002266884, |
| "num_tokens": 4395569.0, |
| "reward": 0.5684863328933716, |
| "reward_std": 0.33382996916770935, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.3715136647224426, |
| "rewards/length_penalty/std": 0.16755269467830658, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9941285848617554, |
| "sampling/importance_sampling_ratio/min": 0.07204371690750122, |
| "sampling/sampling_logp_difference/max": 2.6304821968078613, |
| "sampling/sampling_logp_difference/mean": 0.01957661285996437, |
| "step": 65, |
| "step_time": 23.068237875122577 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005175740210688673, |
| "clip_ratio/high_mean": 0.0005175740210688673, |
| "clip_ratio/low_mean": 0.00023222646414069458, |
| "clip_ratio/low_min": 0.00023222646414069458, |
| "clip_ratio/region_mean": 0.0007498004793887958, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1759.0, |
| "completions/mean_length": 757.97998046875, |
| "completions/mean_terminated_length": 547.9767456054688, |
| "completions/min_length": 294.0, |
| "completions/min_terminated_length": 294.0, |
| "entropy": 0.25827352702617645, |
| "epoch": 0.1793478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016815872862935066, |
| "learning_rate": 5e-05, |
| "loss": 0.036842163652181625, |
| "num_tokens": 4436518.0, |
| "reward": 0.389892578125, |
| "reward_std": 0.6972166895866394, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.37010741233825684, |
| "rewards/length_penalty/std": 0.28814491629600525, |
| "sampling/importance_sampling_ratio/max": 2.3030846118927, |
| "sampling/importance_sampling_ratio/mean": 0.9900518655776978, |
| "sampling/importance_sampling_ratio/min": 0.16431906819343567, |
| "sampling/sampling_logp_difference/max": 1.8059451580047607, |
| "sampling/sampling_logp_difference/mean": 0.026268478482961655, |
| "step": 66, |
| "step_time": 23.87718580919318 |
| }, |
| { |
| "clip_ratio/high_max": 0.00029754412826150657, |
| "clip_ratio/high_mean": 0.00029754412826150657, |
| "clip_ratio/low_mean": 3.082613984588534e-05, |
| "clip_ratio/low_min": 3.082613984588534e-05, |
| "clip_ratio/region_mean": 0.00032837026519700886, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1330.0, |
| "completions/max_terminated_length": 1330.0, |
| "completions/mean_length": 671.2000122070312, |
| "completions/mean_terminated_length": 671.2000122070312, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.10958393663167953, |
| "epoch": 0.18206521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014129486866295338, |
| "learning_rate": 5e-05, |
| "loss": 0.03558261692523956, |
| "num_tokens": 4473028.0, |
| "reward": 0.27226561307907104, |
| "reward_std": 0.5440670847892761, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.3277343809604645, |
| "rewards/length_penalty/std": 0.13625548779964447, |
| "sampling/importance_sampling_ratio/max": 2.361050605773926, |
| "sampling/importance_sampling_ratio/mean": 0.996082603931427, |
| "sampling/importance_sampling_ratio/min": 0.0901251807808876, |
| "sampling/sampling_logp_difference/max": 2.406555652618408, |
| "sampling/sampling_logp_difference/mean": 0.016017161309719086, |
| "step": 67, |
| "step_time": 15.848499092971906 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004814961168449372, |
| "clip_ratio/high_mean": 0.0004814961168449372, |
| "clip_ratio/low_mean": 0.00011306432425044477, |
| "clip_ratio/low_min": 0.00011306432425044477, |
| "clip_ratio/region_mean": 0.000594560446916148, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1327.0, |
| "completions/max_terminated_length": 1327.0, |
| "completions/mean_length": 709.97998046875, |
| "completions/mean_terminated_length": 709.97998046875, |
| "completions/min_length": 377.0, |
| "completions/min_terminated_length": 377.0, |
| "entropy": 0.12251516133546829, |
| "epoch": 0.18478260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017992310225963593, |
| "learning_rate": 5e-05, |
| "loss": 0.034500207751989365, |
| "num_tokens": 4512287.0, |
| "reward": 0.43333005905151367, |
| "reward_std": 0.5297770500183105, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.34666991233825684, |
| "rewards/length_penalty/std": 0.12326952069997787, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9955129027366638, |
| "sampling/importance_sampling_ratio/min": 0.2864378094673157, |
| "sampling/sampling_logp_difference/max": 1.2502338886260986, |
| "sampling/sampling_logp_difference/mean": 0.01752791553735733, |
| "step": 68, |
| "step_time": 16.056356735294685 |
| }, |
| { |
| "clip_ratio/high_max": 0.000786409858847037, |
| "clip_ratio/high_mean": 0.000786409858847037, |
| "clip_ratio/low_mean": 0.00017479720409028233, |
| "clip_ratio/low_min": 0.00017479720409028233, |
| "clip_ratio/region_mean": 0.0009612070745788515, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1685.0, |
| "completions/mean_length": 695.0, |
| "completions/mean_terminated_length": 544.6666870117188, |
| "completions/min_length": 219.0, |
| "completions/min_terminated_length": 219.0, |
| "entropy": 0.2089858502149582, |
| "epoch": 0.1875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01573641039431095, |
| "learning_rate": 5e-05, |
| "loss": 0.06109810620546341, |
| "num_tokens": 4549417.0, |
| "reward": 0.2606445252895355, |
| "reward_std": 0.6902934312820435, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.33935546875, |
| "rewards/length_penalty/std": 0.27389469742774963, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9919227361679077, |
| "sampling/importance_sampling_ratio/min": 0.04992447420954704, |
| "sampling/sampling_logp_difference/max": 2.997243881225586, |
| "sampling/sampling_logp_difference/mean": 0.022705664858222008, |
| "step": 69, |
| "step_time": 24.100224602036178 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003861734687234275, |
| "clip_ratio/high_mean": 0.0003861734687234275, |
| "clip_ratio/low_mean": 8.133393421303481e-05, |
| "clip_ratio/low_min": 8.133393421303481e-05, |
| "clip_ratio/region_mean": 0.00046750739857088776, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1819.0, |
| "completions/mean_length": 974.5999755859375, |
| "completions/mean_terminated_length": 706.25, |
| "completions/min_length": 360.0, |
| "completions/min_terminated_length": 360.0, |
| "entropy": 0.12589940428733826, |
| "epoch": 0.19021739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014950593002140522, |
| "learning_rate": 5e-05, |
| "loss": 0.03755060210824013, |
| "num_tokens": 4601657.0, |
| "reward": -0.27587890625, |
| "reward_std": 0.6061694622039795, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.47587889432907104, |
| "rewards/length_penalty/std": 0.3194302022457123, |
| "sampling/importance_sampling_ratio/max": 2.5902318954467773, |
| "sampling/importance_sampling_ratio/mean": 0.9951319098472595, |
| "sampling/importance_sampling_ratio/min": 0.07221422344446182, |
| "sampling/sampling_logp_difference/max": 2.6281182765960693, |
| "sampling/sampling_logp_difference/mean": 0.015436016023159027, |
| "step": 70, |
| "step_time": 24.753103533759713 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009733674465678632, |
| "clip_ratio/high_mean": 0.0009733674465678632, |
| "clip_ratio/low_mean": 0.00012668214039877058, |
| "clip_ratio/low_min": 0.00012668214039877058, |
| "clip_ratio/region_mean": 0.001100049610249698, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1306.0, |
| "completions/max_terminated_length": 1306.0, |
| "completions/mean_length": 481.8999938964844, |
| "completions/mean_terminated_length": 481.8999938964844, |
| "completions/min_length": 269.0, |
| "completions/min_terminated_length": 269.0, |
| "entropy": 0.18031663894653321, |
| "epoch": 0.19293478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017065592110157013, |
| "learning_rate": 5e-05, |
| "loss": 0.02791469730436802, |
| "num_tokens": 4628112.0, |
| "reward": 0.26469725370407104, |
| "reward_std": 0.5313357710838318, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.23530273139476776, |
| "rewards/length_penalty/std": 0.08668895810842514, |
| "sampling/importance_sampling_ratio/max": 2.9714601039886475, |
| "sampling/importance_sampling_ratio/mean": 0.9928731918334961, |
| "sampling/importance_sampling_ratio/min": 0.06918845325708389, |
| "sampling/sampling_logp_difference/max": 2.6709213256835938, |
| "sampling/sampling_logp_difference/mean": 0.026175597682595253, |
| "step": 71, |
| "step_time": 14.497095804661512 |
| }, |
| { |
| "clip_ratio/high_max": 0.000628325465368107, |
| "clip_ratio/high_mean": 0.000628325465368107, |
| "clip_ratio/low_mean": 4.015257873106748e-05, |
| "clip_ratio/low_min": 4.015257873106748e-05, |
| "clip_ratio/region_mean": 0.0006684780470095575, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1143.0, |
| "completions/max_terminated_length": 1143.0, |
| "completions/mean_length": 483.8999938964844, |
| "completions/mean_terminated_length": 483.8999938964844, |
| "completions/min_length": 178.0, |
| "completions/min_terminated_length": 178.0, |
| "entropy": 0.11079794615507126, |
| "epoch": 0.1956521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01392416376620531, |
| "learning_rate": 5e-05, |
| "loss": 0.02589317597448826, |
| "num_tokens": 4655977.0, |
| "reward": 0.5037207007408142, |
| "reward_std": 0.5148534774780273, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.23627929389476776, |
| "rewards/length_penalty/std": 0.10259860754013062, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9951469898223877, |
| "sampling/importance_sampling_ratio/min": 0.2387021780014038, |
| "sampling/sampling_logp_difference/max": 1.432538628578186, |
| "sampling/sampling_logp_difference/mean": 0.01793760433793068, |
| "step": 72, |
| "step_time": 13.306627204176039 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005853123526321724, |
| "clip_ratio/high_mean": 0.0005853123526321724, |
| "clip_ratio/low_mean": 0.0001358868816168979, |
| "clip_ratio/low_min": 0.0001358868816168979, |
| "clip_ratio/region_mean": 0.0007211992342490703, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2014.0, |
| "completions/mean_length": 723.5999755859375, |
| "completions/mean_terminated_length": 608.434814453125, |
| "completions/min_length": 261.0, |
| "completions/min_terminated_length": 261.0, |
| "entropy": 0.17599567174911498, |
| "epoch": 0.1983695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.018252408131957054, |
| "learning_rate": 5e-05, |
| "loss": 0.04385628551244736, |
| "num_tokens": 4694567.0, |
| "reward": 0.22667968273162842, |
| "reward_std": 0.6738052368164062, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.35332030057907104, |
| "rewards/length_penalty/std": 0.26818665862083435, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9931528568267822, |
| "sampling/importance_sampling_ratio/min": 0.20481857657432556, |
| "sampling/sampling_logp_difference/max": 1.5856306552886963, |
| "sampling/sampling_logp_difference/mean": 0.020317068323493004, |
| "step": 73, |
| "step_time": 24.120045095914975 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005197267164476216, |
| "clip_ratio/high_mean": 0.0005197267164476216, |
| "clip_ratio/low_mean": 7.821666076779366e-05, |
| "clip_ratio/low_min": 7.821666076779366e-05, |
| "clip_ratio/region_mean": 0.0005979433772154152, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 916.0, |
| "completions/max_terminated_length": 916.0, |
| "completions/mean_length": 504.55999755859375, |
| "completions/mean_terminated_length": 504.55999755859375, |
| "completions/min_length": 288.0, |
| "completions/min_terminated_length": 288.0, |
| "entropy": 0.14893011450767518, |
| "epoch": 0.20108695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015567939728498459, |
| "learning_rate": 5e-05, |
| "loss": 0.03300417959690094, |
| "num_tokens": 4723125.0, |
| "reward": 0.15363280475139618, |
| "reward_std": 0.5294827818870544, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.2463671863079071, |
| "rewards/length_penalty/std": 0.09036648273468018, |
| "sampling/importance_sampling_ratio/max": 2.6094131469726562, |
| "sampling/importance_sampling_ratio/mean": 0.9946196675300598, |
| "sampling/importance_sampling_ratio/min": 0.20717699825763702, |
| "sampling/sampling_logp_difference/max": 1.5741817951202393, |
| "sampling/sampling_logp_difference/mean": 0.02010997198522091, |
| "step": 74, |
| "step_time": 11.145676413783804 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009385790792293847, |
| "clip_ratio/high_mean": 0.0009385790792293847, |
| "clip_ratio/low_mean": 0.00018966651987284422, |
| "clip_ratio/low_min": 0.00018966651987284422, |
| "clip_ratio/region_mean": 0.001128245610743761, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1005.0, |
| "completions/max_terminated_length": 1005.0, |
| "completions/mean_length": 429.67999267578125, |
| "completions/mean_terminated_length": 429.67999267578125, |
| "completions/min_length": 153.0, |
| "completions/min_terminated_length": 153.0, |
| "entropy": 0.1344812273979187, |
| "epoch": 0.20380434782608695, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01469445414841175, |
| "learning_rate": 5e-05, |
| "loss": 0.01116997841745615, |
| "num_tokens": 4746939.0, |
| "reward": 0.6301953196525574, |
| "reward_std": 0.38830801844596863, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.2098046839237213, |
| "rewards/length_penalty/std": 0.12929898500442505, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9950018525123596, |
| "sampling/importance_sampling_ratio/min": 0.17992667853832245, |
| "sampling/sampling_logp_difference/max": 1.7152059078216553, |
| "sampling/sampling_logp_difference/mean": 0.022754253819584846, |
| "step": 75, |
| "step_time": 11.691951024346054 |
| }, |
| { |
| "clip_ratio/high_max": 0.00032894672185648235, |
| "clip_ratio/high_mean": 0.00032894672185648235, |
| "clip_ratio/low_mean": 0.0002415040013147518, |
| "clip_ratio/low_min": 0.0002415040013147518, |
| "clip_ratio/region_mean": 0.0005704507231712342, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2009.0, |
| "completions/mean_length": 591.8200073242188, |
| "completions/mean_terminated_length": 430.0222473144531, |
| "completions/min_length": 148.0, |
| "completions/min_terminated_length": 148.0, |
| "entropy": 0.26585462391376496, |
| "epoch": 0.20652173913043478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014526180922985077, |
| "learning_rate": 5e-05, |
| "loss": 0.04992569983005524, |
| "num_tokens": 4779870.0, |
| "reward": 0.4910253882408142, |
| "reward_std": 0.6834759712219238, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.2889746129512787, |
| "rewards/length_penalty/std": 0.2923096716403961, |
| "sampling/importance_sampling_ratio/max": 2.685748338699341, |
| "sampling/importance_sampling_ratio/mean": 0.9892884492874146, |
| "sampling/importance_sampling_ratio/min": 0.30007389187812805, |
| "sampling/sampling_logp_difference/max": 1.2037265300750732, |
| "sampling/sampling_logp_difference/mean": 0.0263529010117054, |
| "step": 76, |
| "step_time": 23.11679401784204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008211379870772361, |
| "clip_ratio/high_mean": 0.0008211379870772361, |
| "clip_ratio/low_mean": 8.669763628859073e-05, |
| "clip_ratio/low_min": 8.669763628859073e-05, |
| "clip_ratio/region_mean": 0.00090783562627621, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1921.0, |
| "completions/mean_length": 535.6799926757812, |
| "completions/mean_terminated_length": 472.66668701171875, |
| "completions/min_length": 186.0, |
| "completions/min_terminated_length": 186.0, |
| "entropy": 0.17671192586421966, |
| "epoch": 0.20923913043478262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01664687879383564, |
| "learning_rate": 5e-05, |
| "loss": 0.04311929643154144, |
| "num_tokens": 4809214.0, |
| "reward": 0.5584374666213989, |
| "reward_std": 0.5280290246009827, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.2615624964237213, |
| "rewards/length_penalty/std": 0.23831947147846222, |
| "sampling/importance_sampling_ratio/max": 2.370129108428955, |
| "sampling/importance_sampling_ratio/mean": 0.9928213953971863, |
| "sampling/importance_sampling_ratio/min": 0.16383084654808044, |
| "sampling/sampling_logp_difference/max": 1.8089207410812378, |
| "sampling/sampling_logp_difference/mean": 0.019636210054159164, |
| "step": 77, |
| "step_time": 22.68614157452248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005915068177273497, |
| "clip_ratio/high_mean": 0.0005915068177273497, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0005915068177273497, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1322.0, |
| "completions/max_terminated_length": 1322.0, |
| "completions/mean_length": 444.0199890136719, |
| "completions/mean_terminated_length": 444.0199890136719, |
| "completions/min_length": 219.0, |
| "completions/min_terminated_length": 219.0, |
| "entropy": 0.09074169397354126, |
| "epoch": 0.21195652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013410658575594425, |
| "learning_rate": 5e-05, |
| "loss": 0.04930553585290909, |
| "num_tokens": 4833715.0, |
| "reward": 0.38319334387779236, |
| "reward_std": 0.5701014995574951, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.21680663526058197, |
| "rewards/length_penalty/std": 0.1205679401755333, |
| "sampling/importance_sampling_ratio/max": 2.962005138397217, |
| "sampling/importance_sampling_ratio/mean": 0.9963091015815735, |
| "sampling/importance_sampling_ratio/min": 0.15177083015441895, |
| "sampling/sampling_logp_difference/max": 1.8853836059570312, |
| "sampling/sampling_logp_difference/mean": 0.014590539038181305, |
| "step": 78, |
| "step_time": 15.235982230165973 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005698791879694909, |
| "clip_ratio/high_mean": 0.0005698791879694909, |
| "clip_ratio/low_mean": 4.1485170368105176e-05, |
| "clip_ratio/low_min": 4.1485170368105176e-05, |
| "clip_ratio/region_mean": 0.0006113643466960639, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 700.0, |
| "completions/max_terminated_length": 700.0, |
| "completions/mean_length": 444.0199890136719, |
| "completions/mean_terminated_length": 444.0199890136719, |
| "completions/min_length": 216.0, |
| "completions/min_terminated_length": 216.0, |
| "entropy": 0.10524996668100357, |
| "epoch": 0.21467391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013324566185474396, |
| "learning_rate": 5e-05, |
| "loss": 0.010624676942825317, |
| "num_tokens": 4859376.0, |
| "reward": 0.5031933188438416, |
| "reward_std": 0.43716707825660706, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.21680663526058197, |
| "rewards/length_penalty/std": 0.06869058310985565, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9948770403862, |
| "sampling/importance_sampling_ratio/min": 0.3475484549999237, |
| "sampling/sampling_logp_difference/max": 1.1924397945404053, |
| "sampling/sampling_logp_difference/mean": 0.019135504961013794, |
| "step": 79, |
| "step_time": 8.87365493294783 |
| }, |
| { |
| "clip_ratio/high_max": 0.00045147799828555436, |
| "clip_ratio/high_mean": 0.00045147799828555436, |
| "clip_ratio/low_mean": 0.0001845527091063559, |
| "clip_ratio/low_min": 0.0001845527091063559, |
| "clip_ratio/region_mean": 0.0006360307103022933, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 665.0, |
| "completions/max_terminated_length": 665.0, |
| "completions/mean_length": 409.3800048828125, |
| "completions/mean_terminated_length": 409.3800048828125, |
| "completions/min_length": 155.0, |
| "completions/min_terminated_length": 155.0, |
| "entropy": 0.11959978342056274, |
| "epoch": 0.21739130434782608, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013123178854584694, |
| "learning_rate": 5e-05, |
| "loss": 0.024036118760704994, |
| "num_tokens": 4883005.0, |
| "reward": 0.2001074105501175, |
| "reward_std": 0.5326245427131653, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.1998925805091858, |
| "rewards/length_penalty/std": 0.05713106691837311, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9956418871879578, |
| "sampling/importance_sampling_ratio/min": 0.3026818335056305, |
| "sampling/sampling_logp_difference/max": 1.1951258182525635, |
| "sampling/sampling_logp_difference/mean": 0.02054966241121292, |
| "step": 80, |
| "step_time": 8.360282597364858 |
| }, |
| { |
| "clip_ratio/high_max": 0.000704864104045555, |
| "clip_ratio/high_mean": 0.000704864104045555, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000704864104045555, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 818.0, |
| "completions/max_terminated_length": 818.0, |
| "completions/mean_length": 417.5799865722656, |
| "completions/mean_terminated_length": 417.5799865722656, |
| "completions/min_length": 214.0, |
| "completions/min_terminated_length": 214.0, |
| "entropy": 0.1103743776679039, |
| "epoch": 0.22010869565217392, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01447275374084711, |
| "learning_rate": 5e-05, |
| "loss": 0.021041925996541977, |
| "num_tokens": 4908344.0, |
| "reward": 0.556103527545929, |
| "reward_std": 0.4081662595272064, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.20389647781848907, |
| "rewards/length_penalty/std": 0.07568074017763138, |
| "sampling/importance_sampling_ratio/max": 2.083041191101074, |
| "sampling/importance_sampling_ratio/mean": 0.9952544569969177, |
| "sampling/importance_sampling_ratio/min": 0.2227083295583725, |
| "sampling/sampling_logp_difference/max": 1.501892328262329, |
| "sampling/sampling_logp_difference/mean": 0.01730303466320038, |
| "step": 81, |
| "step_time": 10.241758018033579 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006575023173354567, |
| "clip_ratio/high_mean": 0.0006575023173354567, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0006575023173354567, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 609.0, |
| "completions/max_terminated_length": 609.0, |
| "completions/mean_length": 294.55999755859375, |
| "completions/mean_terminated_length": 294.55999755859375, |
| "completions/min_length": 119.0, |
| "completions/min_terminated_length": 119.0, |
| "entropy": 0.10942234545946121, |
| "epoch": 0.22282608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012251164764165878, |
| "learning_rate": 5e-05, |
| "loss": 0.009689167141914368, |
| "num_tokens": 4925922.0, |
| "reward": 0.5361718535423279, |
| "reward_std": 0.5191479921340942, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.1438281238079071, |
| "rewards/length_penalty/std": 0.07002057135105133, |
| "sampling/importance_sampling_ratio/max": 2.1058547496795654, |
| "sampling/importance_sampling_ratio/mean": 0.997086763381958, |
| "sampling/importance_sampling_ratio/min": 0.2412642240524292, |
| "sampling/sampling_logp_difference/max": 1.4218626022338867, |
| "sampling/sampling_logp_difference/mean": 0.018847452476620674, |
| "step": 82, |
| "step_time": 7.392252362798899 |
| }, |
| { |
| "clip_ratio/high_max": 0.00035982736153528094, |
| "clip_ratio/high_mean": 0.00035982736153528094, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00035982736153528094, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 491.0, |
| "completions/max_terminated_length": 491.0, |
| "completions/mean_length": 237.3199920654297, |
| "completions/mean_terminated_length": 237.3199920654297, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.1137263223528862, |
| "epoch": 0.22554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012257511727511883, |
| "learning_rate": 5e-05, |
| "loss": 0.013772832229733467, |
| "num_tokens": 4940218.0, |
| "reward": 0.4641210734844208, |
| "reward_std": 0.5065000653266907, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.11587890982627869, |
| "rewards/length_penalty/std": 0.04924225062131882, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9956432580947876, |
| "sampling/importance_sampling_ratio/min": 0.24330933392047882, |
| "sampling/sampling_logp_difference/max": 1.413421630859375, |
| "sampling/sampling_logp_difference/mean": 0.018884913995862007, |
| "step": 83, |
| "step_time": 6.00911526218988 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010932499542832374, |
| "clip_ratio/high_mean": 0.0010932499542832374, |
| "clip_ratio/low_mean": 5.839415825903416e-05, |
| "clip_ratio/low_min": 5.839415825903416e-05, |
| "clip_ratio/region_mean": 0.0011516441009007394, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1617.0, |
| "completions/max_terminated_length": 1617.0, |
| "completions/mean_length": 375.1600036621094, |
| "completions/mean_terminated_length": 375.1600036621094, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.1211256891489029, |
| "epoch": 0.22826086956521738, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019089948385953903, |
| "learning_rate": 5e-05, |
| "loss": 0.03538642078638077, |
| "num_tokens": 4961956.0, |
| "reward": 0.5568163990974426, |
| "reward_std": 0.4542848467826843, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.18318359553813934, |
| "rewards/length_penalty/std": 0.18591263890266418, |
| "sampling/importance_sampling_ratio/max": 2.340315580368042, |
| "sampling/importance_sampling_ratio/mean": 0.995381236076355, |
| "sampling/importance_sampling_ratio/min": 0.24651946127414703, |
| "sampling/sampling_logp_difference/max": 1.4003143310546875, |
| "sampling/sampling_logp_difference/mean": 0.018075764179229736, |
| "step": 84, |
| "step_time": 18.067105296999216 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010971381212584673, |
| "clip_ratio/high_mean": 0.0010971381212584673, |
| "clip_ratio/low_mean": 9.907572530210018e-05, |
| "clip_ratio/low_min": 9.907572530210018e-05, |
| "clip_ratio/region_mean": 0.0011962138232775033, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1556.0, |
| "completions/max_terminated_length": 1556.0, |
| "completions/mean_length": 362.79998779296875, |
| "completions/mean_terminated_length": 362.79998779296875, |
| "completions/min_length": 144.0, |
| "completions/min_terminated_length": 144.0, |
| "entropy": 0.2110247790813446, |
| "epoch": 0.23097826086956522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021158233284950256, |
| "learning_rate": 5e-05, |
| "loss": 0.044072747230529785, |
| "num_tokens": 4982116.0, |
| "reward": 0.3828515410423279, |
| "reward_std": 0.5390759110450745, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.17714843153953552, |
| "rewards/length_penalty/std": 0.12832064926624298, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9906708598136902, |
| "sampling/importance_sampling_ratio/min": 0.171915665268898, |
| "sampling/sampling_logp_difference/max": 1.7607512474060059, |
| "sampling/sampling_logp_difference/mean": 0.028823882341384888, |
| "step": 85, |
| "step_time": 16.47355701122433 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005582125799264759, |
| "clip_ratio/high_mean": 0.0005582125799264759, |
| "clip_ratio/low_mean": 0.00011848341673612595, |
| "clip_ratio/low_min": 0.00011848341673612595, |
| "clip_ratio/region_mean": 0.0006766959966626018, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 332.0, |
| "completions/max_terminated_length": 332.0, |
| "completions/mean_length": 183.47999572753906, |
| "completions/mean_terminated_length": 183.47999572753906, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.14162901043891907, |
| "epoch": 0.23369565217391305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01095115952193737, |
| "learning_rate": 5e-05, |
| "loss": 0.007650045212358236, |
| "num_tokens": 4993390.0, |
| "reward": 0.8504101634025574, |
| "reward_std": 0.24047701060771942, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.08958984166383743, |
| "rewards/length_penalty/std": 0.0338219590485096, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9944058656692505, |
| "sampling/importance_sampling_ratio/min": 0.1472282111644745, |
| "sampling/sampling_logp_difference/max": 1.915771484375, |
| "sampling/sampling_logp_difference/mean": 0.02622547745704651, |
| "step": 86, |
| "step_time": 4.271973532857373 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005829900386743247, |
| "clip_ratio/high_mean": 0.0005829900386743247, |
| "clip_ratio/low_mean": 0.0003810461843386292, |
| "clip_ratio/low_min": 0.0003810461843386292, |
| "clip_ratio/region_mean": 0.0009640362113714218, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 630.0, |
| "completions/max_terminated_length": 630.0, |
| "completions/mean_length": 281.739990234375, |
| "completions/mean_terminated_length": 281.739990234375, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.10393131375312806, |
| "epoch": 0.23641304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013417479582130909, |
| "learning_rate": 5e-05, |
| "loss": 0.014699541963636875, |
| "num_tokens": 5009687.0, |
| "reward": 0.422431617975235, |
| "reward_std": 0.5398931503295898, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.13756835460662842, |
| "rewards/length_penalty/std": 0.07445195317268372, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9958858489990234, |
| "sampling/importance_sampling_ratio/min": 0.13931959867477417, |
| "sampling/sampling_logp_difference/max": 1.970984697341919, |
| "sampling/sampling_logp_difference/mean": 0.018261417746543884, |
| "step": 87, |
| "step_time": 7.5571192188654095 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015571706928312779, |
| "clip_ratio/high_mean": 0.0015571706928312779, |
| "clip_ratio/low_mean": 0.0001285347039811313, |
| "clip_ratio/low_min": 0.0001285347039811313, |
| "clip_ratio/region_mean": 0.0016857054084539413, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 400.0, |
| "completions/max_terminated_length": 400.0, |
| "completions/mean_length": 182.1599884033203, |
| "completions/mean_terminated_length": 182.1599884033203, |
| "completions/min_length": 41.0, |
| "completions/min_terminated_length": 41.0, |
| "entropy": 0.15183552503585815, |
| "epoch": 0.2391304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011707360856235027, |
| "learning_rate": 5e-05, |
| "loss": 0.0038167578168213367, |
| "num_tokens": 5021685.0, |
| "reward": 0.6910547018051147, |
| "reward_std": 0.4340381622314453, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.08894531428813934, |
| "rewards/length_penalty/std": 0.052339497953653336, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9942026138305664, |
| "sampling/importance_sampling_ratio/min": 0.1707821637392044, |
| "sampling/sampling_logp_difference/max": 1.7673664093017578, |
| "sampling/sampling_logp_difference/mean": 0.02865365333855152, |
| "step": 88, |
| "step_time": 5.0701005139853805 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011892693408299237, |
| "clip_ratio/high_mean": 0.0011892693408299237, |
| "clip_ratio/low_mean": 0.000522052450105548, |
| "clip_ratio/low_min": 0.000522052450105548, |
| "clip_ratio/region_mean": 0.0017113217909354717, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 512.0, |
| "completions/max_terminated_length": 512.0, |
| "completions/mean_length": 213.0, |
| "completions/mean_terminated_length": 213.0, |
| "completions/min_length": 63.0, |
| "completions/min_terminated_length": 63.0, |
| "entropy": 0.15271458625793458, |
| "epoch": 0.2418478260869565, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010374904610216618, |
| "learning_rate": 5e-05, |
| "loss": 0.011677139438688755, |
| "num_tokens": 5035795.0, |
| "reward": 0.3359960913658142, |
| "reward_std": 0.5273035168647766, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.10400390625, |
| "rewards/length_penalty/std": 0.061812058091163635, |
| "sampling/importance_sampling_ratio/max": 2.3226864337921143, |
| "sampling/importance_sampling_ratio/mean": 0.993018627166748, |
| "sampling/importance_sampling_ratio/min": 0.08702941983938217, |
| "sampling/sampling_logp_difference/max": 2.4415090084075928, |
| "sampling/sampling_logp_difference/mean": 0.02671627141535282, |
| "step": 89, |
| "step_time": 6.264251670800149 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012449576752260327, |
| "clip_ratio/high_mean": 0.0012449576752260327, |
| "clip_ratio/low_mean": 0.00039986594929359854, |
| "clip_ratio/low_min": 0.00039986594929359854, |
| "clip_ratio/region_mean": 0.001644823612878099, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1107.0, |
| "completions/max_terminated_length": 1107.0, |
| "completions/mean_length": 258.67999267578125, |
| "completions/mean_terminated_length": 258.67999267578125, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.17646301686763763, |
| "epoch": 0.24456521739130435, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013736448250710964, |
| "learning_rate": 5e-05, |
| "loss": 0.030243348330259323, |
| "num_tokens": 5051759.0, |
| "reward": 0.53369140625, |
| "reward_std": 0.510785698890686, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.1263085901737213, |
| "rewards/length_penalty/std": 0.08943063020706177, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9928296804428101, |
| "sampling/importance_sampling_ratio/min": 0.09859311580657959, |
| "sampling/sampling_logp_difference/max": 2.31675386428833, |
| "sampling/sampling_logp_difference/mean": 0.028476905077695847, |
| "step": 90, |
| "step_time": 11.927960007218644 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011836130055598915, |
| "clip_ratio/high_mean": 0.0011836130055598915, |
| "clip_ratio/low_mean": 0.0006012129248119891, |
| "clip_ratio/low_min": 0.0006012129248119891, |
| "clip_ratio/region_mean": 0.0017848259536549448, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 784.0, |
| "completions/max_terminated_length": 784.0, |
| "completions/mean_length": 168.05999755859375, |
| "completions/mean_terminated_length": 168.05999755859375, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.1712634652853012, |
| "epoch": 0.24728260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011951963417232037, |
| "learning_rate": 5e-05, |
| "loss": 0.021161450073122978, |
| "num_tokens": 5065332.0, |
| "reward": 0.5579394698143005, |
| "reward_std": 0.5229561924934387, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.0820605456829071, |
| "rewards/length_penalty/std": 0.08735550940036774, |
| "sampling/importance_sampling_ratio/max": 2.185065507888794, |
| "sampling/importance_sampling_ratio/mean": 0.9940653443336487, |
| "sampling/importance_sampling_ratio/min": 0.12851713597774506, |
| "sampling/sampling_logp_difference/max": 2.0516929626464844, |
| "sampling/sampling_logp_difference/mean": 0.02416444756090641, |
| "step": 91, |
| "step_time": 9.349531583720818 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010452893213368953, |
| "clip_ratio/high_mean": 0.0010452893213368953, |
| "clip_ratio/low_mean": 0.00012650220887735485, |
| "clip_ratio/low_min": 0.00012650220887735485, |
| "clip_ratio/region_mean": 0.00117179153021425, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 292.0, |
| "completions/max_terminated_length": 292.0, |
| "completions/mean_length": 136.17999267578125, |
| "completions/mean_terminated_length": 136.17999267578125, |
| "completions/min_length": 51.0, |
| "completions/min_terminated_length": 51.0, |
| "entropy": 0.15401647984981537, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015128343366086483, |
| "learning_rate": 5e-05, |
| "loss": 0.008885608986020088, |
| "num_tokens": 5074531.0, |
| "reward": 0.5135058760643005, |
| "reward_std": 0.5054919719696045, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.06649413704872131, |
| "rewards/length_penalty/std": 0.029116734862327576, |
| "sampling/importance_sampling_ratio/max": 2.3821372985839844, |
| "sampling/importance_sampling_ratio/mean": 0.9937571287155151, |
| "sampling/importance_sampling_ratio/min": 0.09059903770685196, |
| "sampling/sampling_logp_difference/max": 2.4013116359710693, |
| "sampling/sampling_logp_difference/mean": 0.024070926010608673, |
| "step": 92, |
| "step_time": 4.327101199189201 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013822196982800961, |
| "clip_ratio/high_mean": 0.0013822196982800961, |
| "clip_ratio/low_mean": 0.0005461856024339795, |
| "clip_ratio/low_min": 0.0005461856024339795, |
| "clip_ratio/region_mean": 0.0019284053007140756, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1163.0, |
| "completions/mean_length": 264.32000732421875, |
| "completions/mean_terminated_length": 190.0, |
| "completions/min_length": 20.0, |
| "completions/min_terminated_length": 20.0, |
| "entropy": 0.2661470055580139, |
| "epoch": 0.25271739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020688654854893684, |
| "learning_rate": 5e-05, |
| "loss": 0.10969983786344528, |
| "num_tokens": 5093367.0, |
| "reward": 0.5709375143051147, |
| "reward_std": 0.6100456118583679, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.1290625035762787, |
| "rewards/length_penalty/std": 0.21535643935203552, |
| "sampling/importance_sampling_ratio/max": 2.3409512042999268, |
| "sampling/importance_sampling_ratio/mean": 0.990288496017456, |
| "sampling/importance_sampling_ratio/min": 0.213004007935524, |
| "sampling/sampling_logp_difference/max": 1.546444296836853, |
| "sampling/sampling_logp_difference/mean": 0.02781936153769493, |
| "step": 93, |
| "step_time": 22.407282277243212 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015437419177033007, |
| "clip_ratio/high_mean": 0.0015437419177033007, |
| "clip_ratio/low_mean": 0.00012154360301792621, |
| "clip_ratio/low_min": 0.00012154360301792621, |
| "clip_ratio/region_mean": 0.0016652855207212268, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1413.0, |
| "completions/max_terminated_length": 1413.0, |
| "completions/mean_length": 220.3199920654297, |
| "completions/mean_terminated_length": 220.3199920654297, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.22524437308311462, |
| "epoch": 0.2554347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020771974697709084, |
| "learning_rate": 5e-05, |
| "loss": 0.052876319736242294, |
| "num_tokens": 5110413.0, |
| "reward": 0.412421852350235, |
| "reward_std": 0.5003185272216797, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.10757812857627869, |
| "rewards/length_penalty/std": 0.1135673075914383, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9909684062004089, |
| "sampling/importance_sampling_ratio/min": 0.32960575819015503, |
| "sampling/sampling_logp_difference/max": 1.5696799755096436, |
| "sampling/sampling_logp_difference/mean": 0.028162555769085884, |
| "step": 94, |
| "step_time": 15.704104893608019 |
| }, |
| { |
| "clip_ratio/high_max": 0.001439261925406754, |
| "clip_ratio/high_mean": 0.001439261925406754, |
| "clip_ratio/low_mean": 0.00019417476141825318, |
| "clip_ratio/low_min": 0.00019417476141825318, |
| "clip_ratio/region_mean": 0.0016334366519004107, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 194.0, |
| "completions/max_terminated_length": 194.0, |
| "completions/mean_length": 97.45999908447266, |
| "completions/mean_terminated_length": 97.45999908447266, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.25395846366882324, |
| "epoch": 0.25815217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011895176954567432, |
| "learning_rate": 5e-05, |
| "loss": 0.002930461196228862, |
| "num_tokens": 5117656.0, |
| "reward": 0.6524121165275574, |
| "reward_std": 0.46509990096092224, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.047587890177965164, |
| "rewards/length_penalty/std": 0.019898543134331703, |
| "sampling/importance_sampling_ratio/max": 2.9863996505737305, |
| "sampling/importance_sampling_ratio/mean": 0.9909448623657227, |
| "sampling/importance_sampling_ratio/min": 0.4123409688472748, |
| "sampling/sampling_logp_difference/max": 1.0940685272216797, |
| "sampling/sampling_logp_difference/mean": 0.03773015737533569, |
| "step": 95, |
| "step_time": 2.79798654303886 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014799370197579264, |
| "clip_ratio/high_mean": 0.0014799370197579264, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0014799370197579264, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 551.0, |
| "completions/max_terminated_length": 551.0, |
| "completions/mean_length": 97.69999694824219, |
| "completions/mean_terminated_length": 97.69999694824219, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.20367034077644347, |
| "epoch": 0.2608695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013027592562139034, |
| "learning_rate": 5e-05, |
| "loss": 0.01779348962008953, |
| "num_tokens": 5126071.0, |
| "reward": 0.552294909954071, |
| "reward_std": 0.5121820569038391, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.04770507663488388, |
| "rewards/length_penalty/std": 0.042042821645736694, |
| "sampling/importance_sampling_ratio/max": 2.441518783569336, |
| "sampling/importance_sampling_ratio/mean": 0.991175651550293, |
| "sampling/importance_sampling_ratio/min": 0.06918129324913025, |
| "sampling/sampling_logp_difference/max": 2.671024799346924, |
| "sampling/sampling_logp_difference/mean": 0.03059057705104351, |
| "step": 96, |
| "step_time": 6.202032637782395 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011401220224797725, |
| "clip_ratio/high_mean": 0.0011401220224797725, |
| "clip_ratio/low_mean": 0.0009177979780361056, |
| "clip_ratio/low_min": 0.0009177979780361056, |
| "clip_ratio/region_mean": 0.002057919977232814, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 224.0, |
| "completions/max_terminated_length": 224.0, |
| "completions/mean_length": 86.97999572753906, |
| "completions/mean_terminated_length": 86.97999572753906, |
| "completions/min_length": 29.0, |
| "completions/min_terminated_length": 29.0, |
| "entropy": 0.2265522986650467, |
| "epoch": 0.26358695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01077208761125803, |
| "learning_rate": 5e-05, |
| "loss": 0.0009587033418938518, |
| "num_tokens": 5133350.0, |
| "reward": 0.4375292956829071, |
| "reward_std": 0.5013757944107056, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.04247070476412773, |
| "rewards/length_penalty/std": 0.021450260654091835, |
| "sampling/importance_sampling_ratio/max": 2.6412837505340576, |
| "sampling/importance_sampling_ratio/mean": 0.9908226728439331, |
| "sampling/importance_sampling_ratio/min": 0.3037787675857544, |
| "sampling/sampling_logp_difference/max": 1.191455602645874, |
| "sampling/sampling_logp_difference/mean": 0.033219028264284134, |
| "step": 97, |
| "step_time": 3.065898902947083 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010683529078960418, |
| "clip_ratio/high_mean": 0.0010683529078960418, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010683529078960418, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 150.0, |
| "completions/max_terminated_length": 150.0, |
| "completions/mean_length": 55.55999755859375, |
| "completions/mean_terminated_length": 55.55999755859375, |
| "completions/min_length": 18.0, |
| "completions/min_terminated_length": 18.0, |
| "entropy": 0.2746728152036667, |
| "epoch": 0.266304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012212988920509815, |
| "learning_rate": 5e-05, |
| "loss": 0.007884243503212929, |
| "num_tokens": 5138268.0, |
| "reward": 0.372871071100235, |
| "reward_std": 0.5017322301864624, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.027128906920552254, |
| "rewards/length_penalty/std": 0.014021570794284344, |
| "sampling/importance_sampling_ratio/max": 2.170682907104492, |
| "sampling/importance_sampling_ratio/mean": 0.9889690279960632, |
| "sampling/importance_sampling_ratio/min": 0.35793039202690125, |
| "sampling/sampling_logp_difference/max": 1.0274168252944946, |
| "sampling/sampling_logp_difference/mean": 0.042866162955760956, |
| "step": 98, |
| "step_time": 2.4067599647678435 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008171118563041091, |
| "clip_ratio/high_mean": 0.0008171118563041091, |
| "clip_ratio/low_mean": 0.0007390069542452693, |
| "clip_ratio/low_min": 0.0007390069542452693, |
| "clip_ratio/region_mean": 0.0015561188105493785, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 94.0, |
| "completions/max_terminated_length": 94.0, |
| "completions/mean_length": 50.91999816894531, |
| "completions/mean_terminated_length": 50.91999816894531, |
| "completions/min_length": 22.0, |
| "completions/min_terminated_length": 22.0, |
| "entropy": 0.21310125291347504, |
| "epoch": 0.26902173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011825033463537693, |
| "learning_rate": 5e-05, |
| "loss": 0.0036260976921766996, |
| "num_tokens": 5143724.0, |
| "reward": 0.7351366877555847, |
| "reward_std": 0.4356699287891388, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.02486328035593033, |
| "rewards/length_penalty/std": 0.008906659670174122, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9978911876678467, |
| "sampling/importance_sampling_ratio/min": 0.3821065127849579, |
| "sampling/sampling_logp_difference/max": 1.398585319519043, |
| "sampling/sampling_logp_difference/mean": 0.037688616663217545, |
| "step": 99, |
| "step_time": 2.0239664369728416 |
| }, |
| { |
| "clip_ratio/high_max": 0.0019492614082992077, |
| "clip_ratio/high_mean": 0.0019492614082992077, |
| "clip_ratio/low_mean": 0.0007079645991325378, |
| "clip_ratio/low_min": 0.0007079645991325378, |
| "clip_ratio/region_mean": 0.0026572260074317457, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 125.0, |
| "completions/max_terminated_length": 125.0, |
| "completions/mean_length": 52.87999725341797, |
| "completions/mean_terminated_length": 52.87999725341797, |
| "completions/min_length": 22.0, |
| "completions/min_terminated_length": 22.0, |
| "entropy": 0.22497422993183136, |
| "epoch": 0.2717391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01131928339600563, |
| "learning_rate": 5e-05, |
| "loss": 0.004754737485200167, |
| "num_tokens": 5148598.0, |
| "reward": 0.7741796970367432, |
| "reward_std": 0.4040992259979248, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.025820313021540642, |
| "rewards/length_penalty/std": 0.011249576695263386, |
| "sampling/importance_sampling_ratio/max": 2.4267683029174805, |
| "sampling/importance_sampling_ratio/mean": 0.9914778470993042, |
| "sampling/importance_sampling_ratio/min": 0.27528116106987, |
| "sampling/sampling_logp_difference/max": 1.2899622917175293, |
| "sampling/sampling_logp_difference/mean": 0.03707539662718773, |
| "step": 100, |
| "step_time": 2.2108143279328942 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009703139308840037, |
| "clip_ratio/high_mean": 0.0009703139308840037, |
| "clip_ratio/low_mean": 0.0008246284909546375, |
| "clip_ratio/low_min": 0.0008246284909546375, |
| "clip_ratio/region_mean": 0.0017949424218386412, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 244.0, |
| "completions/max_terminated_length": 244.0, |
| "completions/mean_length": 53.099998474121094, |
| "completions/mean_terminated_length": 53.099998474121094, |
| "completions/min_length": 15.0, |
| "completions/min_terminated_length": 15.0, |
| "entropy": 0.21071316599845885, |
| "epoch": 0.27445652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009538685902953148, |
| "learning_rate": 5e-05, |
| "loss": 0.0037817161064594984, |
| "num_tokens": 5154143.0, |
| "reward": 0.47407224774360657, |
| "reward_std": 0.5069644451141357, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.02592773362994194, |
| "rewards/length_penalty/std": 0.01806822419166565, |
| "sampling/importance_sampling_ratio/max": 2.1848325729370117, |
| "sampling/importance_sampling_ratio/mean": 0.9923020005226135, |
| "sampling/importance_sampling_ratio/min": 0.4618144631385803, |
| "sampling/sampling_logp_difference/max": 0.7815392017364502, |
| "sampling/sampling_logp_difference/mean": 0.032437488436698914, |
| "step": 101, |
| "step_time": 3.6693791849538684 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012555033899843693, |
| "clip_ratio/high_mean": 0.0012555033899843693, |
| "clip_ratio/low_mean": 0.0008362280670553446, |
| "clip_ratio/low_min": 0.0008362280670553446, |
| "clip_ratio/region_mean": 0.0020917314570397137, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 198.0, |
| "completions/max_terminated_length": 198.0, |
| "completions/mean_length": 49.15999984741211, |
| "completions/mean_terminated_length": 49.15999984741211, |
| "completions/min_length": 10.0, |
| "completions/min_terminated_length": 10.0, |
| "entropy": 0.27187140583992003, |
| "epoch": 0.27717391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010691394098103046, |
| "learning_rate": 5e-05, |
| "loss": 0.004616108722984791, |
| "num_tokens": 5159531.0, |
| "reward": 0.5159960985183716, |
| "reward_std": 0.5072278380393982, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.024003906175494194, |
| "rewards/length_penalty/std": 0.018878955394029617, |
| "sampling/importance_sampling_ratio/max": 1.9397785663604736, |
| "sampling/importance_sampling_ratio/mean": 0.9903997182846069, |
| "sampling/importance_sampling_ratio/min": 0.20729371905326843, |
| "sampling/sampling_logp_difference/max": 1.5736185312271118, |
| "sampling/sampling_logp_difference/mean": 0.03977171704173088, |
| "step": 102, |
| "step_time": 2.7362761148251593 |
| }, |
| { |
| "clip_ratio/high_max": 0.000531679327832535, |
| "clip_ratio/high_mean": 0.000531679327832535, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000531679327832535, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1734.0, |
| "completions/max_terminated_length": 1734.0, |
| "completions/mean_length": 133.86000061035156, |
| "completions/mean_terminated_length": 133.86000061035156, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.40329901576042176, |
| "epoch": 0.2798913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01844567432999611, |
| "learning_rate": 5e-05, |
| "loss": 0.03575665503740311, |
| "num_tokens": 5170464.0, |
| "reward": 0.27463865280151367, |
| "reward_std": 0.5309969782829285, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.06536132842302322, |
| "rewards/length_penalty/std": 0.16204895079135895, |
| "sampling/importance_sampling_ratio/max": 2.119347333908081, |
| "sampling/importance_sampling_ratio/mean": 0.9828487634658813, |
| "sampling/importance_sampling_ratio/min": 0.37476417422294617, |
| "sampling/sampling_logp_difference/max": 0.981458306312561, |
| "sampling/sampling_logp_difference/mean": 0.039075400680303574, |
| "step": 103, |
| "step_time": 18.458227863069624 |
| }, |
| { |
| "clip_ratio/high_max": 0.002266806270927191, |
| "clip_ratio/high_mean": 0.002266806270927191, |
| "clip_ratio/low_mean": 0.00035335689317435027, |
| "clip_ratio/low_min": 0.00035335689317435027, |
| "clip_ratio/region_mean": 0.0026201631408184767, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 222.0, |
| "completions/max_terminated_length": 222.0, |
| "completions/mean_length": 51.55999755859375, |
| "completions/mean_terminated_length": 51.55999755859375, |
| "completions/min_length": 13.0, |
| "completions/min_terminated_length": 13.0, |
| "entropy": 0.33404588103294375, |
| "epoch": 0.2826086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013699734583497047, |
| "learning_rate": 5e-05, |
| "loss": 0.011154095642268658, |
| "num_tokens": 5176552.0, |
| "reward": 0.03482421860098839, |
| "reward_std": 0.23582740128040314, |
| "rewards/correctness/mean": 0.05999999865889549, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.025175781920552254, |
| "rewards/length_penalty/std": 0.024918686598539352, |
| "sampling/importance_sampling_ratio/max": 2.478984832763672, |
| "sampling/importance_sampling_ratio/mean": 0.9874392747879028, |
| "sampling/importance_sampling_ratio/min": 0.20078380405902863, |
| "sampling/sampling_logp_difference/max": 1.6055265665054321, |
| "sampling/sampling_logp_difference/mean": 0.04058130085468292, |
| "step": 104, |
| "step_time": 2.9577652697917074 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016553595196455717, |
| "clip_ratio/high_mean": 0.0016553595196455717, |
| "clip_ratio/low_mean": 0.0009756097570061684, |
| "clip_ratio/low_min": 0.0009756097570061684, |
| "clip_ratio/region_mean": 0.00263096927665174, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 65.0, |
| "completions/max_terminated_length": 65.0, |
| "completions/mean_length": 23.059999465942383, |
| "completions/mean_terminated_length": 23.059999465942383, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.21856221556663513, |
| "epoch": 0.28532608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012349529191851616, |
| "learning_rate": 5e-05, |
| "loss": 0.00167317152954638, |
| "num_tokens": 5180065.0, |
| "reward": 0.3287402391433716, |
| "reward_std": 0.4786170721054077, |
| "rewards/correctness/mean": 0.3400000035762787, |
| "rewards/correctness/std": 0.4785180985927582, |
| "rewards/length_penalty/mean": -0.011259765364229679, |
| "rewards/length_penalty/std": 0.005874643102288246, |
| "sampling/importance_sampling_ratio/max": 1.7306820154190063, |
| "sampling/importance_sampling_ratio/mean": 0.9920327067375183, |
| "sampling/importance_sampling_ratio/min": 0.28651663661003113, |
| "sampling/sampling_logp_difference/max": 1.2499587535858154, |
| "sampling/sampling_logp_difference/mean": 0.03913699463009834, |
| "step": 105, |
| "step_time": 1.7049150760285556 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011111111380159855, |
| "clip_ratio/high_mean": 0.0011111111380159855, |
| "clip_ratio/low_mean": 0.0008888889104127883, |
| "clip_ratio/low_min": 0.0008888889104127883, |
| "clip_ratio/region_mean": 0.002000000048428774, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 67.0, |
| "completions/max_terminated_length": 67.0, |
| "completions/mean_length": 19.81999969482422, |
| "completions/mean_terminated_length": 19.81999969482422, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.20343194901943207, |
| "epoch": 0.28804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009475941769778728, |
| "learning_rate": 5e-05, |
| "loss": 0.000543340458534658, |
| "num_tokens": 5183536.0, |
| "reward": 0.37032225728034973, |
| "reward_std": 0.48816052079200745, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.009677734225988388, |
| "rewards/length_penalty/std": 0.005778003018349409, |
| "sampling/importance_sampling_ratio/max": 2.096174955368042, |
| "sampling/importance_sampling_ratio/mean": 0.9992754459381104, |
| "sampling/importance_sampling_ratio/min": 0.2721841633319855, |
| "sampling/sampling_logp_difference/max": 1.301276445388794, |
| "sampling/sampling_logp_difference/mean": 0.03429510071873665, |
| "step": 106, |
| "step_time": 1.7274974128231406 |
| }, |
| { |
| "clip_ratio/high_max": 0.0036198318470269442, |
| "clip_ratio/high_mean": 0.0036198318470269442, |
| "clip_ratio/low_mean": 0.0018731069285422564, |
| "clip_ratio/low_min": 0.0018731069285422564, |
| "clip_ratio/region_mean": 0.0054929387755692, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 137.0, |
| "completions/max_terminated_length": 137.0, |
| "completions/mean_length": 19.479999542236328, |
| "completions/mean_terminated_length": 19.479999542236328, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.27643795013427735, |
| "epoch": 0.2907608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014136921614408493, |
| "learning_rate": 5e-05, |
| "loss": 0.0054918513633310795, |
| "num_tokens": 5188550.0, |
| "reward": 0.1904882788658142, |
| "reward_std": 0.4065392017364502, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.009511718526482582, |
| "rewards/length_penalty/std": 0.00997720006853342, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.993407130241394, |
| "sampling/importance_sampling_ratio/min": 0.28779035806655884, |
| "sampling/sampling_logp_difference/max": 1.2455229759216309, |
| "sampling/sampling_logp_difference/mean": 0.040501918643713, |
| "step": 107, |
| "step_time": 2.2549472115933895 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005633802618831396, |
| "clip_ratio/high_mean": 0.0005633802618831396, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0005633802618831396, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 71.0, |
| "completions/max_terminated_length": 71.0, |
| "completions/mean_length": 26.899999618530273, |
| "completions/mean_terminated_length": 26.899999618530273, |
| "completions/min_length": 10.0, |
| "completions/min_terminated_length": 10.0, |
| "entropy": 0.2395658552646637, |
| "epoch": 0.29347826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013888458721339703, |
| "learning_rate": 5e-05, |
| "loss": 0.004586397670209408, |
| "num_tokens": 5192945.0, |
| "reward": 0.3868652284145355, |
| "reward_std": 0.5013245940208435, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.013134765438735485, |
| "rewards/length_penalty/std": 0.009524103254079819, |
| "sampling/importance_sampling_ratio/max": 1.663638710975647, |
| "sampling/importance_sampling_ratio/mean": 0.9929983615875244, |
| "sampling/importance_sampling_ratio/min": 0.4642603099346161, |
| "sampling/sampling_logp_difference/max": 0.7673099040985107, |
| "sampling/sampling_logp_difference/mean": 0.03226347267627716, |
| "step": 108, |
| "step_time": 1.7853918452747166 |
| }, |
| { |
| "clip_ratio/high_max": 0.0024331796914339064, |
| "clip_ratio/high_mean": 0.0024331796914339064, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0024331796914339064, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 43.0, |
| "completions/max_terminated_length": 43.0, |
| "completions/mean_length": 15.059999465942383, |
| "completions/mean_terminated_length": 15.059999465942383, |
| "completions/min_length": 10.0, |
| "completions/min_terminated_length": 10.0, |
| "entropy": 0.2164798602461815, |
| "epoch": 0.296195652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010043303482234478, |
| "learning_rate": 5e-05, |
| "loss": 0.00015821788110770285, |
| "num_tokens": 5196038.0, |
| "reward": 0.512646496295929, |
| "reward_std": 0.5045772790908813, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.007353515829890966, |
| "rewards/length_penalty/std": 0.0035750912502408028, |
| "sampling/importance_sampling_ratio/max": 1.714004397392273, |
| "sampling/importance_sampling_ratio/mean": 0.9908314347267151, |
| "sampling/importance_sampling_ratio/min": 0.2964578866958618, |
| "sampling/sampling_logp_difference/max": 1.2158501148223877, |
| "sampling/sampling_logp_difference/mean": 0.03589223325252533, |
| "step": 109, |
| "step_time": 1.5638836093712598 |
| }, |
| { |
| "clip_ratio/high_max": 0.002983189094811678, |
| "clip_ratio/high_mean": 0.002983189094811678, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.002983189094811678, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 59.0, |
| "completions/max_terminated_length": 59.0, |
| "completions/mean_length": 18.34000015258789, |
| "completions/mean_terminated_length": 18.34000015258789, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.2114427149295807, |
| "epoch": 0.29891304347826086, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013578943908214569, |
| "learning_rate": 5e-05, |
| "loss": 7.86829914432019e-05, |
| "num_tokens": 5199535.0, |
| "reward": 0.3510449230670929, |
| "reward_std": 0.48104938864707947, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732054233551, |
| "rewards/length_penalty/mean": -0.008955078199505806, |
| "rewards/length_penalty/std": 0.007135857827961445, |
| "sampling/importance_sampling_ratio/max": 1.491443157196045, |
| "sampling/importance_sampling_ratio/mean": 0.9869861006736755, |
| "sampling/importance_sampling_ratio/min": 0.17444418370723724, |
| "sampling/sampling_logp_difference/max": 1.7461504936218262, |
| "sampling/sampling_logp_difference/mean": 0.037049755454063416, |
| "step": 110, |
| "step_time": 1.6775788262020797 |
| }, |
| { |
| "clip_ratio/high_max": 0.006576033774763346, |
| "clip_ratio/high_mean": 0.006576033774763346, |
| "clip_ratio/low_mean": 0.0028456510975956918, |
| "clip_ratio/low_min": 0.0028456510975956918, |
| "clip_ratio/region_mean": 0.009421684872359037, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 44.0, |
| "completions/max_terminated_length": 44.0, |
| "completions/mean_length": 14.679999351501465, |
| "completions/mean_terminated_length": 14.679999351501465, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.12253115922212601, |
| "epoch": 0.3016304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011963631957769394, |
| "learning_rate": 5e-05, |
| "loss": -0.0008805968682281673, |
| "num_tokens": 5202929.0, |
| "reward": 0.5328320264816284, |
| "reward_std": 0.5013919472694397, |
| "rewards/correctness/mean": 0.5400000214576721, |
| "rewards/correctness/std": 0.5034574270248413, |
| "rewards/length_penalty/mean": -0.007167968899011612, |
| "rewards/length_penalty/std": 0.005465332418680191, |
| "sampling/importance_sampling_ratio/max": 2.8982994556427, |
| "sampling/importance_sampling_ratio/mean": 0.9911866188049316, |
| "sampling/importance_sampling_ratio/min": 0.09863191097974777, |
| "sampling/sampling_logp_difference/max": 2.3163604736328125, |
| "sampling/sampling_logp_difference/mean": 0.04423119127750397, |
| "step": 111, |
| "step_time": 1.568905862979591 |
| }, |
| { |
| "clip_ratio/high_max": 0.007079645991325378, |
| "clip_ratio/high_mean": 0.007079645991325378, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.007079645991325378, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 34.0, |
| "completions/max_terminated_length": 34.0, |
| "completions/mean_length": 12.84000015258789, |
| "completions/mean_terminated_length": 12.84000015258789, |
| "completions/min_length": 5.0, |
| "completions/min_terminated_length": 5.0, |
| "entropy": 0.15926921963691712, |
| "epoch": 0.30434782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.025274818763136864, |
| "learning_rate": 5e-05, |
| "loss": 0.0012229093117639422, |
| "num_tokens": 5205871.0, |
| "reward": 0.1937304586172104, |
| "reward_std": 0.40328681468963623, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.006269531324505806, |
| "rewards/length_penalty/std": 0.0021732966415584087, |
| "sampling/importance_sampling_ratio/max": 1.719705581665039, |
| "sampling/importance_sampling_ratio/mean": 0.9994849562644958, |
| "sampling/importance_sampling_ratio/min": 0.5315423607826233, |
| "sampling/sampling_logp_difference/max": 0.6319724321365356, |
| "sampling/sampling_logp_difference/mean": 0.016026850789785385, |
| "step": 112, |
| "step_time": 1.4825945999473333 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0013245033100247384, |
| "clip_ratio/low_min": 0.0013245033100247384, |
| "clip_ratio/region_mean": 0.0013245033100247384, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 28.0, |
| "completions/max_terminated_length": 28.0, |
| "completions/mean_length": 15.819999694824219, |
| "completions/mean_terminated_length": 15.819999694824219, |
| "completions/min_length": 10.0, |
| "completions/min_terminated_length": 10.0, |
| "entropy": 0.13411899358034135, |
| "epoch": 0.3070652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013565887697041035, |
| "learning_rate": 5e-05, |
| "loss": 0.00022481636551674455, |
| "num_tokens": 5209972.0, |
| "reward": 0.1722753942012787, |
| "reward_std": 0.38941752910614014, |
| "rewards/correctness/mean": 0.18000000715255737, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.007724609225988388, |
| "rewards/length_penalty/std": 0.0031237073708325624, |
| "sampling/importance_sampling_ratio/max": 1.3955680131912231, |
| "sampling/importance_sampling_ratio/mean": 0.9914500117301941, |
| "sampling/importance_sampling_ratio/min": 0.0927952229976654, |
| "sampling/sampling_logp_difference/max": 2.3773601055145264, |
| "sampling/sampling_logp_difference/mean": 0.02335107885301113, |
| "step": 113, |
| "step_time": 1.983701549237594 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 10.559999465942383, |
| "completions/mean_terminated_length": 10.559999465942383, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.1528867155313492, |
| "epoch": 0.30978260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020327620208263397, |
| "learning_rate": 5e-05, |
| "loss": -0.00013320728612598032, |
| "num_tokens": 5213040.0, |
| "reward": 0.23484374582767487, |
| "reward_std": 0.4310663938522339, |
| "rewards/correctness/mean": 0.23999999463558197, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.005156250204890966, |
| "rewards/length_penalty/std": 0.0005594291142188013, |
| "sampling/importance_sampling_ratio/max": 1.233178734779358, |
| "sampling/importance_sampling_ratio/mean": 0.9935439229011536, |
| "sampling/importance_sampling_ratio/min": 0.3149997293949127, |
| "sampling/sampling_logp_difference/max": 1.1551835536956787, |
| "sampling/sampling_logp_difference/mean": 0.02408069185912609, |
| "step": 114, |
| "step_time": 1.3785158388782293 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016393441706895827, |
| "clip_ratio/high_mean": 0.0016393441706895827, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0016393441706895827, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 44.0, |
| "completions/max_terminated_length": 44.0, |
| "completions/mean_length": 13.75999927520752, |
| "completions/mean_terminated_length": 13.75999927520752, |
| "completions/min_length": 6.0, |
| "completions/min_terminated_length": 6.0, |
| "entropy": 0.19208596646785736, |
| "epoch": 0.3125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014285330660641193, |
| "learning_rate": 5e-05, |
| "loss": 0.0012205843813717365, |
| "num_tokens": 5215958.0, |
| "reward": 0.1732812523841858, |
| "reward_std": 0.3881458044052124, |
| "rewards/correctness/mean": 0.18000000715255737, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.006718750111758709, |
| "rewards/length_penalty/std": 0.002871919423341751, |
| "sampling/importance_sampling_ratio/max": 2.1067698001861572, |
| "sampling/importance_sampling_ratio/mean": 0.9910176396369934, |
| "sampling/importance_sampling_ratio/min": 0.18786033987998962, |
| "sampling/sampling_logp_difference/max": 1.6720564365386963, |
| "sampling/sampling_logp_difference/mean": 0.035164862871170044, |
| "step": 115, |
| "step_time": 1.5596709728706628 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 29.0, |
| "completions/max_terminated_length": 29.0, |
| "completions/mean_length": 10.179999351501465, |
| "completions/mean_terminated_length": 10.179999351501465, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.09898749366402626, |
| "epoch": 0.31521739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011361517012119293, |
| "learning_rate": 5e-05, |
| "loss": 0.0010609884047880769, |
| "num_tokens": 5219357.0, |
| "reward": 0.3950292766094208, |
| "reward_std": 0.49528878927230835, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.004970703274011612, |
| "rewards/length_penalty/std": 0.001571226166561246, |
| "sampling/importance_sampling_ratio/max": 1.1920545101165771, |
| "sampling/importance_sampling_ratio/mean": 0.9905608296394348, |
| "sampling/importance_sampling_ratio/min": 0.6660861372947693, |
| "sampling/sampling_logp_difference/max": 0.40633630752563477, |
| "sampling/sampling_logp_difference/mean": 0.014935722574591637, |
| "step": 116, |
| "step_time": 1.4674468878656626 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 41.0, |
| "completions/max_terminated_length": 41.0, |
| "completions/mean_length": 11.279999732971191, |
| "completions/mean_terminated_length": 11.279999732971191, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.07753250487148762, |
| "epoch": 0.3179347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021389001980423927, |
| "learning_rate": 5e-05, |
| "loss": 0.0013631616020575166, |
| "num_tokens": 5222101.0, |
| "reward": 0.39449217915534973, |
| "reward_std": 0.4954357147216797, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.0055078123696148396, |
| "rewards/length_penalty/std": 0.0029214380774646997, |
| "sampling/importance_sampling_ratio/max": 1.5011656284332275, |
| "sampling/importance_sampling_ratio/mean": 0.9932869672775269, |
| "sampling/importance_sampling_ratio/min": 0.22313006222248077, |
| "sampling/sampling_logp_difference/max": 1.5000004768371582, |
| "sampling/sampling_logp_difference/mean": 0.02301364578306675, |
| "step": 117, |
| "step_time": 1.4666734847705811 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010204081423580646, |
| "clip_ratio/high_mean": 0.0010204081423580646, |
| "clip_ratio/low_mean": 0.0025129454210400582, |
| "clip_ratio/low_min": 0.0025129454210400582, |
| "clip_ratio/region_mean": 0.003533353563398123, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 85.0, |
| "completions/max_terminated_length": 85.0, |
| "completions/mean_length": 17.34000015258789, |
| "completions/mean_terminated_length": 17.34000015258789, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.15508841276168822, |
| "epoch": 0.32065217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01637706533074379, |
| "learning_rate": 5e-05, |
| "loss": 0.0006858168635517359, |
| "num_tokens": 5226648.0, |
| "reward": 0.3515332043170929, |
| "reward_std": 0.4793729782104492, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.008466796949505806, |
| "rewards/length_penalty/std": 0.007427203468978405, |
| "sampling/importance_sampling_ratio/max": 2.323932409286499, |
| "sampling/importance_sampling_ratio/mean": 0.9894140362739563, |
| "sampling/importance_sampling_ratio/min": 0.2500162422657013, |
| "sampling/sampling_logp_difference/max": 1.386229395866394, |
| "sampling/sampling_logp_difference/mean": 0.03073703683912754, |
| "step": 118, |
| "step_time": 1.8715046828147024 |
| }, |
| { |
| "clip_ratio/high_max": 0.001431485399371013, |
| "clip_ratio/high_mean": 0.001431485399371013, |
| "clip_ratio/low_mean": 0.0021437447983771564, |
| "clip_ratio/low_min": 0.0021437447983771564, |
| "clip_ratio/region_mean": 0.0035752301977481694, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 77.0, |
| "completions/mean_length": 65.26000213623047, |
| "completions/mean_terminated_length": 24.795917510986328, |
| "completions/min_length": 5.0, |
| "completions/min_terminated_length": 5.0, |
| "entropy": 0.17566183917224407, |
| "epoch": 0.3233695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013526733964681625, |
| "learning_rate": 5e-05, |
| "loss": 0.08021794259548187, |
| "num_tokens": 5233421.0, |
| "reward": 0.3281347453594208, |
| "reward_std": 0.5138391852378845, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.031865235418081284, |
| "rewards/length_penalty/std": 0.14017029106616974, |
| "sampling/importance_sampling_ratio/max": 1.5155799388885498, |
| "sampling/importance_sampling_ratio/mean": 0.9970149397850037, |
| "sampling/importance_sampling_ratio/min": 0.48622196912765503, |
| "sampling/sampling_logp_difference/max": 0.7210900187492371, |
| "sampling/sampling_logp_difference/mean": 0.009188398718833923, |
| "step": 119, |
| "step_time": 20.337825877359137 |
| }, |
| { |
| "clip_ratio/high_max": 0.002040816284716129, |
| "clip_ratio/high_mean": 0.002040816284716129, |
| "clip_ratio/low_mean": 0.0021739130839705466, |
| "clip_ratio/low_min": 0.0021739130839705466, |
| "clip_ratio/region_mean": 0.004214729368686676, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 18.0, |
| "completions/max_terminated_length": 18.0, |
| "completions/mean_length": 10.380000114440918, |
| "completions/mean_terminated_length": 10.380000114440918, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.25275845229625704, |
| "epoch": 0.32608695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010559819638729095, |
| "learning_rate": 5e-05, |
| "loss": 0.00029482506215572357, |
| "num_tokens": 5236790.0, |
| "reward": 0.07493163645267487, |
| "reward_std": 0.27439793944358826, |
| "rewards/correctness/mean": 0.07999999821186066, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.0050683594308793545, |
| "rewards/length_penalty/std": 0.0015687467530369759, |
| "sampling/importance_sampling_ratio/max": 1.5421497821807861, |
| "sampling/importance_sampling_ratio/mean": 0.9877521991729736, |
| "sampling/importance_sampling_ratio/min": 0.38020059466362, |
| "sampling/sampling_logp_difference/max": 0.9670562744140625, |
| "sampling/sampling_logp_difference/mean": 0.03305759280920029, |
| "step": 120, |
| "step_time": 1.3726711603812873 |
| }, |
| { |
| "clip_ratio/high_max": 0.004678522609174252, |
| "clip_ratio/high_mean": 0.004678522609174252, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.004678522609174252, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 22.0, |
| "completions/max_terminated_length": 22.0, |
| "completions/mean_length": 9.319999694824219, |
| "completions/mean_terminated_length": 9.319999694824219, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.06333770230412483, |
| "epoch": 0.328804347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.036573126912117004, |
| "learning_rate": 5e-05, |
| "loss": 0.0010053031146526337, |
| "num_tokens": 5240596.0, |
| "reward": 0.3954492211341858, |
| "reward_std": 0.49500203132629395, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.004550781100988388, |
| "rewards/length_penalty/std": 0.0012298959773033857, |
| "sampling/importance_sampling_ratio/max": 1.5860633850097656, |
| "sampling/importance_sampling_ratio/mean": 0.9944274425506592, |
| "sampling/importance_sampling_ratio/min": 0.5976189970970154, |
| "sampling/sampling_logp_difference/max": 0.5148018598556519, |
| "sampling/sampling_logp_difference/mean": 0.011470520868897438, |
| "step": 121, |
| "step_time": 1.4338487619534135 |
| }, |
| { |
| "clip_ratio/high_max": 0.0021276595070958138, |
| "clip_ratio/high_mean": 0.0021276595070958138, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0021276595070958138, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 10.09999942779541, |
| "completions/mean_terminated_length": 10.09999942779541, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.1215081237256527, |
| "epoch": 0.33152173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024983931332826614, |
| "learning_rate": 5e-05, |
| "loss": 0.0001665241434238851, |
| "num_tokens": 5244221.0, |
| "reward": 0.1350683569908142, |
| "reward_std": 0.350132554769516, |
| "rewards/correctness/mean": 0.14000000059604645, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.004931640811264515, |
| "rewards/length_penalty/std": 0.0008614040561951697, |
| "sampling/importance_sampling_ratio/max": 2.7834041118621826, |
| "sampling/importance_sampling_ratio/mean": 0.9930852651596069, |
| "sampling/importance_sampling_ratio/min": 0.43255409598350525, |
| "sampling/sampling_logp_difference/max": 1.023674726486206, |
| "sampling/sampling_logp_difference/mean": 0.02821943163871765, |
| "step": 122, |
| "step_time": 1.373616496566683 |
| }, |
| { |
| "clip_ratio/high_max": 0.0035325242206454277, |
| "clip_ratio/high_mean": 0.0035325242206454277, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0035325242206454277, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 17.0, |
| "completions/max_terminated_length": 17.0, |
| "completions/mean_length": 11.5, |
| "completions/mean_terminated_length": 11.5, |
| "completions/min_length": 5.0, |
| "completions/min_terminated_length": 5.0, |
| "entropy": 0.12081824392080306, |
| "epoch": 0.3342391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019469941034913063, |
| "learning_rate": 5e-05, |
| "loss": 0.0006497701397165656, |
| "num_tokens": 5248036.0, |
| "reward": -0.005615234375, |
| "reward_std": 0.001525853993371129, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.005615234375, |
| "rewards/length_penalty/std": 0.001525853993371129, |
| "sampling/importance_sampling_ratio/max": 1.4670969247817993, |
| "sampling/importance_sampling_ratio/mean": 0.9967119693756104, |
| "sampling/importance_sampling_ratio/min": 0.07796137034893036, |
| "sampling/sampling_logp_difference/max": 2.551541805267334, |
| "sampling/sampling_logp_difference/mean": 0.03180255368351936, |
| "step": 123, |
| "step_time": 1.416173886274919 |
| }, |
| { |
| "clip_ratio/high_max": 0.009455096907913685, |
| "clip_ratio/high_mean": 0.009455096907913685, |
| "clip_ratio/low_mean": 0.0034365782514214514, |
| "clip_ratio/low_min": 0.0034365782514214514, |
| "clip_ratio/region_mean": 0.012891675159335137, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 20.0, |
| "completions/max_terminated_length": 20.0, |
| "completions/mean_length": 10.859999656677246, |
| "completions/mean_terminated_length": 10.859999656677246, |
| "completions/min_length": 5.0, |
| "completions/min_terminated_length": 5.0, |
| "entropy": 0.11404812037944793, |
| "epoch": 0.33695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024839911609888077, |
| "learning_rate": 5e-05, |
| "loss": -0.0001064792595570907, |
| "num_tokens": 5250829.0, |
| "reward": 0.45469725131988525, |
| "reward_std": 0.5023077726364136, |
| "rewards/correctness/mean": 0.46000000834465027, |
| "rewards/correctness/std": 0.5034573674201965, |
| "rewards/length_penalty/mean": -0.005302734207361937, |
| "rewards/length_penalty/std": 0.0019974883180111647, |
| "sampling/importance_sampling_ratio/max": 1.6004278659820557, |
| "sampling/importance_sampling_ratio/mean": 0.9722920656204224, |
| "sampling/importance_sampling_ratio/min": 0.01483890786767006, |
| "sampling/sampling_logp_difference/max": 4.210502624511719, |
| "sampling/sampling_logp_difference/mean": 0.08302152901887894, |
| "step": 124, |
| "step_time": 1.3816925161518157 |
| }, |
| { |
| "clip_ratio/high_max": 0.0033976585022173823, |
| "clip_ratio/high_mean": 0.0033976585022173823, |
| "clip_ratio/low_mean": 0.002684563770890236, |
| "clip_ratio/low_min": 0.002684563770890236, |
| "clip_ratio/region_mean": 0.006082222273107618, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 63.0, |
| "completions/mean_length": 52.39999771118164, |
| "completions/mean_terminated_length": 11.673469543457031, |
| "completions/min_length": 5.0, |
| "completions/min_terminated_length": 5.0, |
| "entropy": 0.06118136048316956, |
| "epoch": 0.33967391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014367754571139812, |
| "learning_rate": 5e-05, |
| "loss": 0.07726050913333893, |
| "num_tokens": 5255929.0, |
| "reward": 0.3744140565395355, |
| "reward_std": 0.5283631682395935, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.02558593824505806, |
| "rewards/length_penalty/std": 0.1407146453857422, |
| "sampling/importance_sampling_ratio/max": 2.106475353240967, |
| "sampling/importance_sampling_ratio/mean": 0.9985984563827515, |
| "sampling/importance_sampling_ratio/min": 0.3653067648410797, |
| "sampling/sampling_logp_difference/max": 1.0070178508758545, |
| "sampling/sampling_logp_difference/mean": 0.004823117051273584, |
| "step": 125, |
| "step_time": 20.179652540944517 |
| }, |
| { |
| "clip_ratio/high_max": 0.00827543493360281, |
| "clip_ratio/high_mean": 0.00827543493360281, |
| "clip_ratio/low_mean": 0.002083333395421505, |
| "clip_ratio/low_min": 0.002083333395421505, |
| "clip_ratio/region_mean": 0.010358768329024316, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14.0, |
| "completions/max_terminated_length": 14.0, |
| "completions/mean_length": 10.199999809265137, |
| "completions/mean_terminated_length": 10.199999809265137, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.1312358409166336, |
| "epoch": 0.3423913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010240713134407997, |
| "learning_rate": 5e-05, |
| "loss": 9.463776223128662e-05, |
| "num_tokens": 5261409.0, |
| "reward": 0.05501953139901161, |
| "reward_std": 0.23984263837337494, |
| "rewards/correctness/mean": 0.05999999865889549, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.0049804686568677425, |
| "rewards/length_penalty/std": 0.0011713765561580658, |
| "sampling/importance_sampling_ratio/max": 2.973083257675171, |
| "sampling/importance_sampling_ratio/mean": 0.9844138622283936, |
| "sampling/importance_sampling_ratio/min": 0.2322712540626526, |
| "sampling/sampling_logp_difference/max": 1.4598493576049805, |
| "sampling/sampling_logp_difference/mean": 0.05247528478503227, |
| "step": 126, |
| "step_time": 1.426745121134445 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015625, |
| "clip_ratio/high_mean": 0.0015625, |
| "clip_ratio/low_mean": 0.010037288442254067, |
| "clip_ratio/low_min": 0.010037288442254067, |
| "clip_ratio/region_mean": 0.011599788442254066, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 18.0, |
| "completions/max_terminated_length": 18.0, |
| "completions/mean_length": 11.519999504089355, |
| "completions/mean_terminated_length": 11.519999504089355, |
| "completions/min_length": 6.0, |
| "completions/min_terminated_length": 6.0, |
| "entropy": 0.12240712642669678, |
| "epoch": 0.3451086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01998254843056202, |
| "learning_rate": 5e-05, |
| "loss": 0.0006313637131825089, |
| "num_tokens": 5264615.0, |
| "reward": -0.005624999757856131, |
| "reward_std": 0.001768079586327076, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.005625000223517418, |
| "rewards/length_penalty/std": 0.001768079586327076, |
| "sampling/importance_sampling_ratio/max": 1.4942423105239868, |
| "sampling/importance_sampling_ratio/mean": 0.9801560640335083, |
| "sampling/importance_sampling_ratio/min": 0.02733716182410717, |
| "sampling/sampling_logp_difference/max": 3.599508285522461, |
| "sampling/sampling_logp_difference/mean": 0.06355856359004974, |
| "step": 127, |
| "step_time": 1.4069360962603241 |
| }, |
| { |
| "clip_ratio/high_max": 0.005801975913345814, |
| "clip_ratio/high_mean": 0.005801975913345814, |
| "clip_ratio/low_mean": 0.008586328104138374, |
| "clip_ratio/low_min": 0.008586328104138374, |
| "clip_ratio/region_mean": 0.014388304017484189, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10.0, |
| "completions/max_terminated_length": 10.0, |
| "completions/mean_length": 7.21999979019165, |
| "completions/mean_terminated_length": 7.21999979019165, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.14161121100187302, |
| "epoch": 0.34782608695652173, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016605541110038757, |
| "learning_rate": 5e-05, |
| "loss": 0.0001308184873778373, |
| "num_tokens": 5267296.0, |
| "reward": 0.056474607437849045, |
| "reward_std": 0.23955321311950684, |
| "rewards/correctness/mean": 0.05999999865889549, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.003525390522554517, |
| "rewards/length_penalty/std": 0.0009680049843154848, |
| "sampling/importance_sampling_ratio/max": 2.0141804218292236, |
| "sampling/importance_sampling_ratio/mean": 0.9972620010375977, |
| "sampling/importance_sampling_ratio/min": 0.2654344439506531, |
| "sampling/sampling_logp_difference/max": 1.3263874053955078, |
| "sampling/sampling_logp_difference/mean": 0.04654447361826897, |
| "step": 128, |
| "step_time": 1.34115265798755 |
| }, |
| { |
| "clip_ratio/high_max": 0.0052564961835741995, |
| "clip_ratio/high_mean": 0.0052564961835741995, |
| "clip_ratio/low_mean": 0.0030814639292657377, |
| "clip_ratio/low_min": 0.0030814639292657377, |
| "clip_ratio/region_mean": 0.008337960112839937, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 29.0, |
| "completions/max_terminated_length": 29.0, |
| "completions/mean_length": 11.960000038146973, |
| "completions/mean_terminated_length": 11.960000038146973, |
| "completions/min_length": 8.0, |
| "completions/min_terminated_length": 8.0, |
| "entropy": 0.145098540186882, |
| "epoch": 0.35054347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019570035859942436, |
| "learning_rate": 5e-05, |
| "loss": 0.001352812978439033, |
| "num_tokens": 5270954.0, |
| "reward": -0.0058398437686264515, |
| "reward_std": 0.0026634151581674814, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0058398437686264515, |
| "rewards/length_penalty/std": 0.0026634151581674814, |
| "sampling/importance_sampling_ratio/max": 2.7703750133514404, |
| "sampling/importance_sampling_ratio/mean": 0.9858039021492004, |
| "sampling/importance_sampling_ratio/min": 0.4572809338569641, |
| "sampling/sampling_logp_difference/max": 1.0189826488494873, |
| "sampling/sampling_logp_difference/mean": 0.03138452768325806, |
| "step": 129, |
| "step_time": 1.4406253728084266 |
| }, |
| { |
| "clip_ratio/high_max": 0.023539825342595577, |
| "clip_ratio/high_mean": 0.023539825342595577, |
| "clip_ratio/low_mean": 0.01050705425441265, |
| "clip_ratio/low_min": 0.01050705425441265, |
| "clip_ratio/region_mean": 0.03404687978327274, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10.0, |
| "completions/max_terminated_length": 10.0, |
| "completions/mean_length": 7.159999847412109, |
| "completions/mean_terminated_length": 7.159999847412109, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.28884335458278654, |
| "epoch": 0.3532608695652174, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024998050183057785, |
| "learning_rate": 5e-05, |
| "loss": 0.0004798858717549592, |
| "num_tokens": 5274082.0, |
| "reward": 0.3765038847923279, |
| "reward_std": 0.4897673428058624, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.003496093675494194, |
| "rewards/length_penalty/std": 0.0010867377277463675, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9742715954780579, |
| "sampling/importance_sampling_ratio/min": 0.09587825834751129, |
| "sampling/sampling_logp_difference/max": 2.3446760177612305, |
| "sampling/sampling_logp_difference/mean": 0.13332518935203552, |
| "step": 130, |
| "step_time": 1.328675802797079 |
| }, |
| { |
| "clip_ratio/high_max": 0.003103244863450527, |
| "clip_ratio/high_mean": 0.003103244863450527, |
| "clip_ratio/low_mean": 0.0033828146755695344, |
| "clip_ratio/low_min": 0.0033828146755695344, |
| "clip_ratio/region_mean": 0.006486059539020062, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 62.0, |
| "completions/max_terminated_length": 62.0, |
| "completions/mean_length": 9.539999961853027, |
| "completions/mean_terminated_length": 9.539999961853027, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.23320847153663635, |
| "epoch": 0.35597826086956524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017455505207180977, |
| "learning_rate": 5e-05, |
| "loss": 0.003064260818064213, |
| "num_tokens": 5277939.0, |
| "reward": -0.0046582031063735485, |
| "reward_std": 0.006479442585259676, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0046582031063735485, |
| "rewards/length_penalty/std": 0.006479442585259676, |
| "sampling/importance_sampling_ratio/max": 1.6986380815505981, |
| "sampling/importance_sampling_ratio/mean": 0.9788010716438293, |
| "sampling/importance_sampling_ratio/min": 0.15788955986499786, |
| "sampling/sampling_logp_difference/max": 1.8458595275878906, |
| "sampling/sampling_logp_difference/mean": 0.05766844004392624, |
| "step": 131, |
| "step_time": 1.6266542652156204 |
| }, |
| { |
| "clip_ratio/high_max": 9.46073792874813e-05, |
| "clip_ratio/high_mean": 9.46073792874813e-05, |
| "clip_ratio/low_mean": 0.00634920671582222, |
| "clip_ratio/low_min": 0.00634920671582222, |
| "clip_ratio/region_mean": 0.006443814095109701, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 13.0, |
| "completions/mean_length": 47.20000076293945, |
| "completions/mean_terminated_length": 6.36734676361084, |
| "completions/min_length": 4.0, |
| "completions/min_terminated_length": 4.0, |
| "entropy": 0.20117006599903106, |
| "epoch": 0.358695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021256737411022186, |
| "learning_rate": 5e-05, |
| "loss": 0.060951780527830124, |
| "num_tokens": 5282129.0, |
| "reward": -0.003046874888241291, |
| "reward_std": 0.20168958604335785, |
| "rewards/correctness/mean": 0.019999999552965164, |
| "rewards/correctness/std": 0.1414213627576828, |
| "rewards/length_penalty/mean": -0.02304687537252903, |
| "rewards/length_penalty/std": 0.14098763465881348, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9934961199760437, |
| "sampling/importance_sampling_ratio/min": 3.782039357247413e-07, |
| "sampling/sampling_logp_difference/max": 14.787832260131836, |
| "sampling/sampling_logp_difference/mean": 0.04299402981996536, |
| "step": 132, |
| "step_time": 19.68206503125839 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5.0, |
| "completions/max_terminated_length": 5.0, |
| "completions/mean_length": 4.199999809265137, |
| "completions/mean_terminated_length": 4.199999809265137, |
| "completions/min_length": 3.0, |
| "completions/min_terminated_length": 3.0, |
| "entropy": 0.23295619785785676, |
| "epoch": 0.36141304347826086, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5285029.0, |
| "reward": 0.19794921576976776, |
| "reward_std": 0.40435701608657837, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.0020507811568677425, |
| "rewards/length_penalty/std": 0.00036910592461936176, |
| "sampling/importance_sampling_ratio/max": 1.188761591911316, |
| "sampling/importance_sampling_ratio/mean": 0.9596621990203857, |
| "sampling/importance_sampling_ratio/min": 0.024618901312351227, |
| "sampling/sampling_logp_difference/max": 3.7042407989501953, |
| "sampling/sampling_logp_difference/mean": 0.12237008661031723, |
| "step": 133, |
| "step_time": 1.264468238921836 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0054054055362939835, |
| "clip_ratio/low_min": 0.0054054055362939835, |
| "clip_ratio/region_mean": 0.0054054055362939835, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6.0, |
| "completions/max_terminated_length": 6.0, |
| "completions/mean_length": 3.319999933242798, |
| "completions/mean_terminated_length": 3.319999933242798, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.17164091616868973, |
| "epoch": 0.3641304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006110444199293852, |
| "learning_rate": 5e-05, |
| "loss": 0.00015725757111795247, |
| "num_tokens": 5287885.0, |
| "reward": 0.19837890565395355, |
| "reward_std": 0.4043866693973541, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.0016210937174037099, |
| "rewards/length_penalty/std": 0.00043494580313563347, |
| "sampling/importance_sampling_ratio/max": 1.356533169746399, |
| "sampling/importance_sampling_ratio/mean": 0.9950489401817322, |
| "sampling/importance_sampling_ratio/min": 0.0008358496124856174, |
| "sampling/sampling_logp_difference/max": 7.087061882019043, |
| "sampling/sampling_logp_difference/mean": 0.07953914254903793, |
| "step": 134, |
| "step_time": 1.3009774491656572 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0035087719559669496, |
| "clip_ratio/low_min": 0.0035087719559669496, |
| "clip_ratio/region_mean": 0.0035087719559669496, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 4.599999904632568, |
| "completions/mean_terminated_length": 4.599999904632568, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.2973464459180832, |
| "epoch": 0.36684782608695654, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011810951866209507, |
| "learning_rate": 5e-05, |
| "loss": 0.0003894786350429058, |
| "num_tokens": 5292295.0, |
| "reward": -0.0022460937034338713, |
| "reward_std": 0.0009665461839176714, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0022460937034338713, |
| "rewards/length_penalty/std": 0.0009665461839176714, |
| "sampling/importance_sampling_ratio/max": 1.6216562986373901, |
| "sampling/importance_sampling_ratio/mean": 0.9891418218612671, |
| "sampling/importance_sampling_ratio/min": 0.11668897420167923, |
| "sampling/sampling_logp_difference/max": 2.1482431888580322, |
| "sampling/sampling_logp_difference/mean": 0.07455261051654816, |
| "step": 135, |
| "step_time": 1.3720859428867698 |
| }, |
| { |
| "clip_ratio/high_max": 0.01212609987705946, |
| "clip_ratio/high_mean": 0.01212609987705946, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.01212609987705946, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 107.0, |
| "completions/max_terminated_length": 107.0, |
| "completions/mean_length": 6.420000076293945, |
| "completions/mean_terminated_length": 6.420000076293945, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.4212904691696167, |
| "epoch": 0.3695652173913043, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009215766564011574, |
| "learning_rate": 5e-05, |
| "loss": 0.0023378136102110147, |
| "num_tokens": 5294696.0, |
| "reward": 0.21686522662639618, |
| "reward_std": 0.419146329164505, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.003134765662252903, |
| "rewards/length_penalty/std": 0.0071464055217802525, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9935191869735718, |
| "sampling/importance_sampling_ratio/min": 0.30233317613601685, |
| "sampling/sampling_logp_difference/max": 1.2943968772888184, |
| "sampling/sampling_logp_difference/mean": 0.08518271148204803, |
| "step": 136, |
| "step_time": 1.852066189981997 |
| }, |
| { |
| "clip_ratio/high_max": 0.00416666679084301, |
| "clip_ratio/high_mean": 0.00416666679084301, |
| "clip_ratio/low_mean": 0.016872549057006837, |
| "clip_ratio/low_min": 0.016872549057006837, |
| "clip_ratio/region_mean": 0.021039215847849846, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 26.0, |
| "completions/max_terminated_length": 26.0, |
| "completions/mean_length": 6.71999979019165, |
| "completions/mean_terminated_length": 6.71999979019165, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.2728432804346085, |
| "epoch": 0.37228260869565216, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.027852702885866165, |
| "learning_rate": 5e-05, |
| "loss": 0.0009244568645954132, |
| "num_tokens": 5297892.0, |
| "reward": 0.3567187488079071, |
| "reward_std": 0.4856407046318054, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.003281249897554517, |
| "rewards/length_penalty/std": 0.0028727324679493904, |
| "sampling/importance_sampling_ratio/max": 1.9078514575958252, |
| "sampling/importance_sampling_ratio/mean": 0.9902607202529907, |
| "sampling/importance_sampling_ratio/min": 0.32021790742874146, |
| "sampling/sampling_logp_difference/max": 1.1387535333633423, |
| "sampling/sampling_logp_difference/mean": 0.07158994674682617, |
| "step": 137, |
| "step_time": 1.4134835042059422 |
| }, |
| { |
| "clip_ratio/high_max": 0.016225016117095946, |
| "clip_ratio/high_mean": 0.016225016117095946, |
| "clip_ratio/low_mean": 0.00416666679084301, |
| "clip_ratio/low_min": 0.00416666679084301, |
| "clip_ratio/region_mean": 0.020391682535409926, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15.0, |
| "completions/max_terminated_length": 15.0, |
| "completions/mean_length": 5.079999923706055, |
| "completions/mean_terminated_length": 5.079999923706055, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.17912932634353637, |
| "epoch": 0.375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017889760434627533, |
| "learning_rate": 5e-05, |
| "loss": 0.00015456334222108126, |
| "num_tokens": 5301176.0, |
| "reward": 0.15751953423023224, |
| "reward_std": 0.3685983121395111, |
| "rewards/correctness/mean": 0.1599999964237213, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.002480468712747097, |
| "rewards/length_penalty/std": 0.0021268592681735754, |
| "sampling/importance_sampling_ratio/max": 1.55757737159729, |
| "sampling/importance_sampling_ratio/mean": 0.9653849005699158, |
| "sampling/importance_sampling_ratio/min": 0.3918115496635437, |
| "sampling/sampling_logp_difference/max": 0.936974287033081, |
| "sampling/sampling_logp_difference/mean": 0.05680634453892708, |
| "step": 138, |
| "step_time": 1.783718638587743 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 2.5, |
| "completions/mean_terminated_length": 2.5, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.11814753264188767, |
| "epoch": 0.37771739130434784, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01109712291508913, |
| "learning_rate": 5e-05, |
| "loss": 3.969172394135967e-05, |
| "num_tokens": 5304081.0, |
| "reward": 0.2187792956829071, |
| "reward_std": 0.4180828928947449, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.001220703125, |
| "rewards/length_penalty/std": 0.0004214232030790299, |
| "sampling/importance_sampling_ratio/max": 1.2242869138717651, |
| "sampling/importance_sampling_ratio/mean": 0.9961508512496948, |
| "sampling/importance_sampling_ratio/min": 0.16473731398582458, |
| "sampling/sampling_logp_difference/max": 1.8034031391143799, |
| "sampling/sampling_logp_difference/mean": 0.055593159049749374, |
| "step": 139, |
| "step_time": 1.2756988110486418 |
| }, |
| { |
| "clip_ratio/high_max": 0.00625, |
| "clip_ratio/high_mean": 0.00625, |
| "clip_ratio/low_mean": 0.01653299927711487, |
| "clip_ratio/low_min": 0.01653299927711487, |
| "clip_ratio/region_mean": 0.022782999277114867, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5.0, |
| "completions/max_terminated_length": 5.0, |
| "completions/mean_length": 3.5399999618530273, |
| "completions/mean_terminated_length": 3.5399999618530273, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.21788379549980164, |
| "epoch": 0.3804347826086957, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014283841475844383, |
| "learning_rate": 5e-05, |
| "loss": 7.27055812603794e-05, |
| "num_tokens": 5307898.0, |
| "reward": 0.01827148348093033, |
| "reward_std": 0.14146029949188232, |
| "rewards/correctness/mean": 0.019999999552965164, |
| "rewards/correctness/std": 0.1414213627576828, |
| "rewards/length_penalty/mean": -0.0017285156063735485, |
| "rewards/length_penalty/std": 0.0005050338804721832, |
| "sampling/importance_sampling_ratio/max": 2.1249887943267822, |
| "sampling/importance_sampling_ratio/mean": 0.980110228061676, |
| "sampling/importance_sampling_ratio/min": 0.34859681129455566, |
| "sampling/sampling_logp_difference/max": 1.0538392066955566, |
| "sampling/sampling_logp_difference/mean": 0.04972858726978302, |
| "step": 140, |
| "step_time": 1.3094815532676876 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4.0, |
| "completions/max_terminated_length": 4.0, |
| "completions/mean_length": 3.419999837875366, |
| "completions/mean_terminated_length": 3.419999837875366, |
| "completions/min_length": 3.0, |
| "completions/min_terminated_length": 3.0, |
| "entropy": 0.030646063201129437, |
| "epoch": 0.38315217391304346, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004051357973366976, |
| "learning_rate": 5e-05, |
| "loss": 2.7182053599972278e-05, |
| "num_tokens": 5310799.0, |
| "reward": 0.39833006262779236, |
| "reward_std": 0.4946380853652954, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.001669921912252903, |
| "rewards/length_penalty/std": 0.0002434420894132927, |
| "sampling/importance_sampling_ratio/max": 1.028769850730896, |
| "sampling/importance_sampling_ratio/mean": 0.9956688284873962, |
| "sampling/importance_sampling_ratio/min": 0.4431457221508026, |
| "sampling/sampling_logp_difference/max": 0.8138566017150879, |
| "sampling/sampling_logp_difference/mean": 0.008249584585428238, |
| "step": 141, |
| "step_time": 1.2636189230252057 |
| }, |
| { |
| "clip_ratio/high_max": 0.006451612710952759, |
| "clip_ratio/high_mean": 0.006451612710952759, |
| "clip_ratio/low_mean": 0.013118279725313186, |
| "clip_ratio/low_min": 0.013118279725313186, |
| "clip_ratio/region_mean": 0.019569892436265945, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5.0, |
| "completions/max_terminated_length": 5.0, |
| "completions/mean_length": 3.179999828338623, |
| "completions/mean_terminated_length": 3.179999828338623, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.41223096251487734, |
| "epoch": 0.3858695652173913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007143351249396801, |
| "learning_rate": 5e-05, |
| "loss": -4.87293436890468e-05, |
| "num_tokens": 5313748.0, |
| "reward": -0.0015527342911809683, |
| "reward_std": 0.0005640199524350464, |
| "rewards/correctness/mean": 0.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.0015527344075962901, |
| "rewards/length_penalty/std": 0.0005640199524350464, |
| "sampling/importance_sampling_ratio/max": 2.4648897647857666, |
| "sampling/importance_sampling_ratio/mean": 0.9650440812110901, |
| "sampling/importance_sampling_ratio/min": 0.28366824984550476, |
| "sampling/sampling_logp_difference/max": 1.259949803352356, |
| "sampling/sampling_logp_difference/mean": 0.11519324779510498, |
| "step": 142, |
| "step_time": 1.2917137211188674 |
| }, |
| { |
| "clip_ratio/high_max": 0.003999999910593033, |
| "clip_ratio/high_mean": 0.003999999910593033, |
| "clip_ratio/low_mean": 0.003999999910593033, |
| "clip_ratio/low_min": 0.003999999910593033, |
| "clip_ratio/region_mean": 0.007999999821186066, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 4.119999885559082, |
| "completions/mean_terminated_length": 4.119999885559082, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.14669495075941086, |
| "epoch": 0.38858695652173914, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03668024763464928, |
| "learning_rate": 5e-05, |
| "loss": -1.1713493222487159e-05, |
| "num_tokens": 5316294.0, |
| "reward": 0.4979882836341858, |
| "reward_std": 0.50434809923172, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.0020117186941206455, |
| "rewards/length_penalty/std": 0.0015334563795477152, |
| "sampling/importance_sampling_ratio/max": 1.1682599782943726, |
| "sampling/importance_sampling_ratio/mean": 0.9931513071060181, |
| "sampling/importance_sampling_ratio/min": 0.3147142827510834, |
| "sampling/sampling_logp_difference/max": 1.1560900211334229, |
| "sampling/sampling_logp_difference/mean": 0.03275330364704132, |
| "step": 143, |
| "step_time": 1.3322547457646579 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12.0, |
| "completions/max_terminated_length": 12.0, |
| "completions/mean_length": 4.400000095367432, |
| "completions/mean_terminated_length": 4.400000095367432, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.28117292523384096, |
| "epoch": 0.391304347826087, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014653222635388374, |
| "learning_rate": 5e-05, |
| "loss": 0.0007023264188319445, |
| "num_tokens": 5319564.0, |
| "reward": 0.19785155355930328, |
| "reward_std": 0.4046550989151001, |
| "rewards/correctness/mean": 0.20000000298023224, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.0021484375465661287, |
| "rewards/length_penalty/std": 0.0014631820376962423, |
| "sampling/importance_sampling_ratio/max": 1.436386227607727, |
| "sampling/importance_sampling_ratio/mean": 0.982090175151825, |
| "sampling/importance_sampling_ratio/min": 0.49318453669548035, |
| "sampling/sampling_logp_difference/max": 0.7068718671798706, |
| "sampling/sampling_logp_difference/mean": 0.03635406866669655, |
| "step": 144, |
| "step_time": 1.3282462656497955 |
| }, |
| { |
| "clip_ratio/high_max": 0.00555555559694767, |
| "clip_ratio/high_mean": 0.00555555559694767, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00555555559694767, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7.0, |
| "completions/max_terminated_length": 7.0, |
| "completions/mean_length": 4.019999980926514, |
| "completions/mean_terminated_length": 4.019999980926514, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.06284114345908165, |
| "epoch": 0.39402173913043476, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009140221402049065, |
| "learning_rate": 5e-05, |
| "loss": 7.00859964126721e-05, |
| "num_tokens": 5322095.0, |
| "reward": 0.238037109375, |
| "reward_std": 0.43184083700180054, |
| "rewards/correctness/mean": 0.23999999463558197, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.0019628906156867743, |
| "rewards/length_penalty/std": 0.0007147025316953659, |
| "sampling/importance_sampling_ratio/max": 1.8910298347473145, |
| "sampling/importance_sampling_ratio/mean": 0.9930315017700195, |
| "sampling/importance_sampling_ratio/min": 0.3770080804824829, |
| "sampling/sampling_logp_difference/max": 0.9754886627197266, |
| "sampling/sampling_logp_difference/mean": 0.022765684872865677, |
| "step": 145, |
| "step_time": 1.3034854554571211 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.008695652335882187, |
| "clip_ratio/low_min": 0.008695652335882187, |
| "clip_ratio/region_mean": 0.008695652335882187, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3.0, |
| "completions/max_terminated_length": 3.0, |
| "completions/mean_length": 2.240000009536743, |
| "completions/mean_terminated_length": 2.240000009536743, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.18227319419384003, |
| "epoch": 0.3967391304347826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02299347147345543, |
| "learning_rate": 5e-05, |
| "loss": 4.629989416571334e-05, |
| "num_tokens": 5325147.0, |
| "reward": 0.39890623092651367, |
| "reward_std": 0.4947669804096222, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487170577049255, |
| "rewards/length_penalty/mean": -0.0010937500046566129, |
| "rewards/length_penalty/std": 0.0002106538595398888, |
| "sampling/importance_sampling_ratio/max": 1.1442818641662598, |
| "sampling/importance_sampling_ratio/mean": 1.0006992816925049, |
| "sampling/importance_sampling_ratio/min": 0.5129154324531555, |
| "sampling/sampling_logp_difference/max": 0.6676442623138428, |
| "sampling/sampling_logp_difference/mean": 0.0328700952231884, |
| "step": 146, |
| "step_time": 1.2859444648493081 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8.0, |
| "completions/max_terminated_length": 8.0, |
| "completions/mean_length": 3.1999998092651367, |
| "completions/mean_terminated_length": 3.1999998092651367, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.043758167326450347, |
| "epoch": 0.39945652173913043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 5327607.0, |
| "reward": 0.3984375, |
| "reward_std": 0.4941476285457611, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.0015625000232830644, |
| "rewards/length_penalty/std": 0.0011837725760415196, |
| "sampling/importance_sampling_ratio/max": 1.0770561695098877, |
| "sampling/importance_sampling_ratio/mean": 0.9971351623535156, |
| "sampling/importance_sampling_ratio/min": 0.36504772305488586, |
| "sampling/sampling_logp_difference/max": 1.0077271461486816, |
| "sampling/sampling_logp_difference/mean": 0.016550671309232712, |
| "step": 147, |
| "step_time": 1.3252897900529206 |
| }, |
| { |
| "clip_ratio/high_max": 0.007999999821186066, |
| "clip_ratio/high_mean": 0.007999999821186066, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.007999999821186066, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3.0, |
| "completions/max_terminated_length": 3.0, |
| "completions/mean_length": 2.5999999046325684, |
| "completions/mean_terminated_length": 2.5999999046325684, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.24429004788398742, |
| "epoch": 0.40217391304347827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.025154076516628265, |
| "learning_rate": 5e-05, |
| "loss": 7.58409732952714e-05, |
| "num_tokens": 5330647.0, |
| "reward": 0.15873046219348907, |
| "reward_std": 0.37024199962615967, |
| "rewards/correctness/mean": 0.1599999964237213, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.0012695312034338713, |
| "rewards/length_penalty/std": 0.00024163654597941786, |
| "sampling/importance_sampling_ratio/max": 1.84139084815979, |
| "sampling/importance_sampling_ratio/mean": 1.0035784244537354, |
| "sampling/importance_sampling_ratio/min": 0.2585758864879608, |
| "sampling/sampling_logp_difference/max": 1.352566123008728, |
| "sampling/sampling_logp_difference/mean": 0.08900177478790283, |
| "step": 148, |
| "step_time": 1.2713856103364378 |
| }, |
| { |
| "clip_ratio/high_max": 0.004894911497831345, |
| "clip_ratio/high_mean": 0.004894911497831345, |
| "clip_ratio/low_mean": 0.00786290317773819, |
| "clip_ratio/low_min": 0.00786290317773819, |
| "clip_ratio/region_mean": 0.012757814675569534, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 57.0, |
| "completions/max_terminated_length": 57.0, |
| "completions/mean_length": 9.779999732971191, |
| "completions/mean_terminated_length": 9.779999732971191, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.3053502798080444, |
| "epoch": 0.4048913043478261, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021597005426883698, |
| "learning_rate": 5e-05, |
| "loss": 0.0015805705916136503, |
| "num_tokens": 5334106.0, |
| "reward": 0.21522460877895355, |
| "reward_std": 0.4204765558242798, |
| "rewards/correctness/mean": 0.2199999988079071, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.0047753904946148396, |
| "rewards/length_penalty/std": 0.007258681580424309, |
| "sampling/importance_sampling_ratio/max": 1.8426822423934937, |
| "sampling/importance_sampling_ratio/mean": 0.9776626825332642, |
| "sampling/importance_sampling_ratio/min": 0.38441887497901917, |
| "sampling/sampling_logp_difference/max": 0.9560225009918213, |
| "sampling/sampling_logp_difference/mean": 0.0729394257068634, |
| "step": 149, |
| "step_time": 1.6369031788781285 |
| }, |
| { |
| "clip_ratio/high_max": 0.006666667014360428, |
| "clip_ratio/high_mean": 0.006666667014360428, |
| "clip_ratio/low_mean": 0.05519783571362495, |
| "clip_ratio/low_min": 0.05519783571362495, |
| "clip_ratio/region_mean": 0.061864501982927325, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5.0, |
| "completions/max_terminated_length": 5.0, |
| "completions/mean_length": 2.879999876022339, |
| "completions/mean_terminated_length": 2.879999876022339, |
| "completions/min_length": 2.0, |
| "completions/min_terminated_length": 2.0, |
| "entropy": 0.16436495780944824, |
| "epoch": 0.4076086956521739, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021375233307480812, |
| "learning_rate": 5e-05, |
| "loss": 0.00015874237578827888, |
| "num_tokens": 5336010.0, |
| "reward": 0.5785937309265137, |
| "reward_std": 0.49848005175590515, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.0014062500558793545, |
| "rewards/length_penalty/std": 0.00046938066952861845, |
| "sampling/importance_sampling_ratio/max": 1.5337551832199097, |
| "sampling/importance_sampling_ratio/mean": 0.9424799084663391, |
| "sampling/importance_sampling_ratio/min": 0.12197969108819962, |
| "sampling/sampling_logp_difference/max": 2.103900671005249, |
| "sampling/sampling_logp_difference/mean": 0.1282735913991928, |
| "step": 150, |
| "step_time": 1.6516870979685336 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 200, |
| "num_input_tokens_seen": 5336010, |
| "num_train_epochs": 1, |
| "save_steps": 50, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|