| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5597014925373134, |
| "eval_steps": 500, |
| "global_step": 150, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 567.0, |
| "completions/max_terminated_length": 567.0, |
| "completions/mean_length": 259.2799987792969, |
| "completions/mean_terminated_length": 259.2799987792969, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.1668988436460495, |
| "epoch": 0.0037313432835820895, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006947502493858337, |
| "learning_rate": 0.0, |
| "loss": 0.0044, |
| "num_tokens": 15194.0, |
| "reward": 0.7933984398841858, |
| "reward_std": 0.29082080721855164, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.12660156190395355, |
| "rewards/length_penalty/std": 0.04607566446065903, |
| "sampling/importance_sampling_ratio/max": 1.3554484844207764, |
| "sampling/importance_sampling_ratio/mean": 0.9939242601394653, |
| "sampling/importance_sampling_ratio/min": 0.6691522598266602, |
| "sampling/sampling_logp_difference/max": 0.40174365043640137, |
| "sampling/sampling_logp_difference/mean": 0.011276595294475555, |
| "step": 1, |
| "step_time": 6.9493025781121105 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1107.0, |
| "completions/max_terminated_length": 1107.0, |
| "completions/mean_length": 412.0, |
| "completions/mean_terminated_length": 412.0, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.1244592621922493, |
| "epoch": 0.007462686567164179, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007026078645139933, |
| "learning_rate": 5e-06, |
| "loss": 0.0366, |
| "num_tokens": 38684.0, |
| "reward": 0.3988281190395355, |
| "reward_std": 0.5786729454994202, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.201171875, |
| "rewards/length_penalty/std": 0.12524312734603882, |
| "sampling/importance_sampling_ratio/max": 1.4278866052627563, |
| "sampling/importance_sampling_ratio/mean": 0.9958950281143188, |
| "sampling/importance_sampling_ratio/min": 0.6846740245819092, |
| "sampling/sampling_logp_difference/max": 0.37881243228912354, |
| "sampling/sampling_logp_difference/mean": 0.008306741714477539, |
| "step": 2, |
| "step_time": 12.598701566690579 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007442941830959171, |
| "clip_ratio/high_mean": 0.0007442941830959171, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007442941830959171, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 672.0, |
| "completions/max_terminated_length": 672.0, |
| "completions/mean_length": 306.5799865722656, |
| "completions/mean_terminated_length": 306.5799865722656, |
| "completions/min_length": 188.0, |
| "completions/min_terminated_length": 188.0, |
| "entropy": 0.16790179610252381, |
| "epoch": 0.011194029850746268, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008647753857076168, |
| "learning_rate": 1e-05, |
| "loss": 0.0019, |
| "num_tokens": 56933.0, |
| "reward": 0.4303027391433716, |
| "reward_std": 0.4748076796531677, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.14969725906848907, |
| "rewards/length_penalty/std": 0.05218725651502609, |
| "sampling/importance_sampling_ratio/max": 1.413362741470337, |
| "sampling/importance_sampling_ratio/mean": 0.993857741355896, |
| "sampling/importance_sampling_ratio/min": 0.6936461925506592, |
| "sampling/sampling_logp_difference/max": 0.36579322814941406, |
| "sampling/sampling_logp_difference/mean": 0.011682353913784027, |
| "step": 3, |
| "step_time": 8.017057088203728 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006428720662370324, |
| "clip_ratio/high_mean": 0.0006428720662370324, |
| "clip_ratio/low_mean": 4.096681659575552e-05, |
| "clip_ratio/low_min": 4.096681659575552e-05, |
| "clip_ratio/region_mean": 0.000683838885743171, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1016.0, |
| "completions/max_terminated_length": 1016.0, |
| "completions/mean_length": 445.1600036621094, |
| "completions/mean_terminated_length": 445.1600036621094, |
| "completions/min_length": 205.0, |
| "completions/min_terminated_length": 205.0, |
| "entropy": 0.12619452476501464, |
| "epoch": 0.014925373134328358, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007980461232364178, |
| "learning_rate": 1.5e-05, |
| "loss": 0.0268, |
| "num_tokens": 82271.0, |
| "reward": 0.5826367139816284, |
| "reward_std": 0.5049587488174438, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.21736328303813934, |
| "rewards/length_penalty/std": 0.11474549770355225, |
| "sampling/importance_sampling_ratio/max": 1.430497407913208, |
| "sampling/importance_sampling_ratio/mean": 0.9957796931266785, |
| "sampling/importance_sampling_ratio/min": 0.7457529306411743, |
| "sampling/sampling_logp_difference/max": 0.35802221298217773, |
| "sampling/sampling_logp_difference/mean": 0.00873645767569542, |
| "step": 4, |
| "step_time": 12.020404138602316 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011642194353044034, |
| "clip_ratio/high_mean": 0.0011642194353044034, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011642194353044034, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 601.0, |
| "completions/max_terminated_length": 601.0, |
| "completions/mean_length": 290.8800048828125, |
| "completions/mean_terminated_length": 290.8800048828125, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.1662202298641205, |
| "epoch": 0.018656716417910446, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005730130709707737, |
| "learning_rate": 2e-05, |
| "loss": 0.0008, |
| "num_tokens": 99835.0, |
| "reward": 0.8379687070846558, |
| "reward_std": 0.15626150369644165, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1420312523841858, |
| "rewards/length_penalty/std": 0.06337719410657883, |
| "sampling/importance_sampling_ratio/max": 1.3366026878356934, |
| "sampling/importance_sampling_ratio/mean": 0.9940405488014221, |
| "sampling/importance_sampling_ratio/min": 0.7202601432800293, |
| "sampling/sampling_logp_difference/max": 0.3281428813934326, |
| "sampling/sampling_logp_difference/mean": 0.011546244844794273, |
| "step": 5, |
| "step_time": 7.627342459047213 |
| }, |
| { |
| "clip_ratio/high_max": 0.000865393309504725, |
| "clip_ratio/high_mean": 0.000865393309504725, |
| "clip_ratio/low_mean": 4.660918202716857e-05, |
| "clip_ratio/low_min": 4.660918202716857e-05, |
| "clip_ratio/region_mean": 0.0009120024915318936, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1064.0, |
| "completions/max_terminated_length": 1064.0, |
| "completions/mean_length": 485.53997802734375, |
| "completions/mean_terminated_length": 485.53997802734375, |
| "completions/min_length": 220.0, |
| "completions/min_terminated_length": 220.0, |
| "entropy": 0.14179427921772003, |
| "epoch": 0.022388059701492536, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008097657933831215, |
| "learning_rate": 2.5e-05, |
| "loss": -0.0103, |
| "num_tokens": 127112.0, |
| "reward": 0.7229198813438416, |
| "reward_std": 0.240312397480011, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.23708008229732513, |
| "rewards/length_penalty/std": 0.11706869304180145, |
| "sampling/importance_sampling_ratio/max": 1.4403629302978516, |
| "sampling/importance_sampling_ratio/mean": 0.9951878190040588, |
| "sampling/importance_sampling_ratio/min": 0.6609635949134827, |
| "sampling/sampling_logp_difference/max": 0.41405653953552246, |
| "sampling/sampling_logp_difference/mean": 0.009446932002902031, |
| "step": 6, |
| "step_time": 12.106442798860371 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007034160080365837, |
| "clip_ratio/high_mean": 0.0007034160080365837, |
| "clip_ratio/low_mean": 5.22593007190153e-05, |
| "clip_ratio/low_min": 5.22593007190153e-05, |
| "clip_ratio/region_mean": 0.0007556752883829176, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 686.5599975585938, |
| "completions/mean_terminated_length": 599.6595458984375, |
| "completions/min_length": 212.0, |
| "completions/min_terminated_length": 212.0, |
| "entropy": 0.2536372452974319, |
| "epoch": 0.026119402985074626, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00851096399128437, |
| "learning_rate": 3e-05, |
| "loss": 0.0075, |
| "num_tokens": 165200.0, |
| "reward": 0.36476561427116394, |
| "reward_std": 0.7040084004402161, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.3352343738079071, |
| "rewards/length_penalty/std": 0.2852962613105774, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9906913638114929, |
| "sampling/importance_sampling_ratio/min": 0.676866352558136, |
| "sampling/sampling_logp_difference/max": 1.6592447757720947, |
| "sampling/sampling_logp_difference/mean": 0.016186492517590523, |
| "step": 7, |
| "step_time": 22.643113143742085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004955741052981466, |
| "clip_ratio/high_mean": 0.0004955741052981466, |
| "clip_ratio/low_mean": 0.00013108916173223406, |
| "clip_ratio/low_min": 0.00013108916173223406, |
| "clip_ratio/region_mean": 0.0006266632699407637, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1593.0, |
| "completions/mean_length": 585.8999633789062, |
| "completions/mean_terminated_length": 492.574462890625, |
| "completions/min_length": 210.0, |
| "completions/min_terminated_length": 210.0, |
| "entropy": 0.22597861886024476, |
| "epoch": 0.029850746268656716, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015066474676132202, |
| "learning_rate": 3.5e-05, |
| "loss": 0.1228, |
| "num_tokens": 198345.0, |
| "reward": 0.19391600787639618, |
| "reward_std": 0.6122970581054688, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.28608399629592896, |
| "rewards/length_penalty/std": 0.23632247745990753, |
| "sampling/importance_sampling_ratio/max": 1.5942074060440063, |
| "sampling/importance_sampling_ratio/mean": 0.992152214050293, |
| "sampling/importance_sampling_ratio/min": 0.6727555394172668, |
| "sampling/sampling_logp_difference/max": 0.4663766622543335, |
| "sampling/sampling_logp_difference/mean": 0.014843679964542389, |
| "step": 8, |
| "step_time": 22.47531992616132 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006101833190768957, |
| "clip_ratio/high_mean": 0.0006101833190768957, |
| "clip_ratio/low_mean": 5.7159189600497486e-05, |
| "clip_ratio/low_min": 5.7159189600497486e-05, |
| "clip_ratio/region_mean": 0.0006673425086773932, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 587.0, |
| "completions/max_terminated_length": 587.0, |
| "completions/mean_length": 336.1399841308594, |
| "completions/mean_terminated_length": 336.1399841308594, |
| "completions/min_length": 216.0, |
| "completions/min_terminated_length": 216.0, |
| "entropy": 0.12334485203027726, |
| "epoch": 0.033582089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004527137149125338, |
| "learning_rate": 4e-05, |
| "loss": 0.0018, |
| "num_tokens": 217742.0, |
| "reward": 0.8158690929412842, |
| "reward_std": 0.14488673210144043, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.16413086652755737, |
| "rewards/length_penalty/std": 0.04868381842970848, |
| "sampling/importance_sampling_ratio/max": 1.4266942739486694, |
| "sampling/importance_sampling_ratio/mean": 0.9956211447715759, |
| "sampling/importance_sampling_ratio/min": 0.7008862495422363, |
| "sampling/sampling_logp_difference/max": 0.3554096221923828, |
| "sampling/sampling_logp_difference/mean": 0.008958252146840096, |
| "step": 9, |
| "step_time": 7.467382221948355 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002761587995337322, |
| "clip_ratio/high_mean": 0.0002761587995337322, |
| "clip_ratio/low_mean": 0.00011805738904513419, |
| "clip_ratio/low_min": 0.00011805738904513419, |
| "clip_ratio/region_mean": 0.0003942161885788664, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1656.0, |
| "completions/mean_length": 615.239990234375, |
| "completions/mean_terminated_length": 555.5416870117188, |
| "completions/min_length": 118.0, |
| "completions/min_terminated_length": 118.0, |
| "entropy": 0.1781733125448227, |
| "epoch": 0.03731343283582089, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00930685643106699, |
| "learning_rate": 4.5e-05, |
| "loss": 0.0254, |
| "num_tokens": 252474.0, |
| "reward": 0.2995898425579071, |
| "reward_std": 0.6082340478897095, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.3004101514816284, |
| "rewards/length_penalty/std": 0.23450656235218048, |
| "sampling/importance_sampling_ratio/max": 1.450623631477356, |
| "sampling/importance_sampling_ratio/mean": 0.9936089515686035, |
| "sampling/importance_sampling_ratio/min": 0.2059124857187271, |
| "sampling/sampling_logp_difference/max": 1.5803040266036987, |
| "sampling/sampling_logp_difference/mean": 0.011203057132661343, |
| "step": 10, |
| "step_time": 22.559994000708684 |
| }, |
| { |
| "clip_ratio/high_max": 0.000726152554852888, |
| "clip_ratio/high_mean": 0.000726152554852888, |
| "clip_ratio/low_mean": 0.00017200752045027913, |
| "clip_ratio/low_min": 0.00017200752045027913, |
| "clip_ratio/region_mean": 0.0008981600753031671, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 712.0, |
| "completions/max_terminated_length": 712.0, |
| "completions/mean_length": 309.0799865722656, |
| "completions/mean_terminated_length": 309.0799865722656, |
| "completions/min_length": 112.0, |
| "completions/min_terminated_length": 112.0, |
| "entropy": 0.12753532230854034, |
| "epoch": 0.041044776119402986, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005573405418545008, |
| "learning_rate": 5e-05, |
| "loss": -0.0067, |
| "num_tokens": 270938.0, |
| "reward": 0.4090820252895355, |
| "reward_std": 0.5463029742240906, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.15091796219348907, |
| "rewards/length_penalty/std": 0.08516758680343628, |
| "sampling/importance_sampling_ratio/max": 1.4622268676757812, |
| "sampling/importance_sampling_ratio/mean": 0.9957054853439331, |
| "sampling/importance_sampling_ratio/min": 0.6586275696754456, |
| "sampling/sampling_logp_difference/max": 0.41759705543518066, |
| "sampling/sampling_logp_difference/mean": 0.008718845434486866, |
| "step": 11, |
| "step_time": 8.197410132037476 |
| }, |
| { |
| "clip_ratio/high_max": 0.001248956541530788, |
| "clip_ratio/high_mean": 0.001248956541530788, |
| "clip_ratio/low_mean": 0.0002312510390765965, |
| "clip_ratio/low_min": 0.0002312510390765965, |
| "clip_ratio/region_mean": 0.001480207545682788, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1557.0, |
| "completions/mean_length": 540.8999633789062, |
| "completions/mean_terminated_length": 510.1428527832031, |
| "completions/min_length": 236.0, |
| "completions/min_terminated_length": 236.0, |
| "entropy": 0.24297742843627929, |
| "epoch": 0.04477611940298507, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011518114246428013, |
| "learning_rate": 5e-05, |
| "loss": -0.0239, |
| "num_tokens": 301663.0, |
| "reward": 0.535888671875, |
| "reward_std": 0.4949640929698944, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.26411134004592896, |
| "rewards/length_penalty/std": 0.18795593082904816, |
| "sampling/importance_sampling_ratio/max": 1.5822714567184448, |
| "sampling/importance_sampling_ratio/mean": 0.9911604523658752, |
| "sampling/importance_sampling_ratio/min": 0.6676426529884338, |
| "sampling/sampling_logp_difference/max": 0.45886147022247314, |
| "sampling/sampling_logp_difference/mean": 0.015835359692573547, |
| "step": 12, |
| "step_time": 21.649674186250195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007544187363237142, |
| "clip_ratio/high_mean": 0.0007544187363237142, |
| "clip_ratio/low_mean": 0.00023230386723298578, |
| "clip_ratio/low_min": 0.00023230386723298578, |
| "clip_ratio/region_mean": 0.0009867226035567, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1960.0, |
| "completions/mean_length": 761.8999633789062, |
| "completions/mean_terminated_length": 516.9285888671875, |
| "completions/min_length": 102.0, |
| "completions/min_terminated_length": 102.0, |
| "entropy": 0.3580703973770142, |
| "epoch": 0.048507462686567165, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011552696116268635, |
| "learning_rate": 5e-05, |
| "loss": 0.0463, |
| "num_tokens": 343388.0, |
| "reward": 0.22797851264476776, |
| "reward_std": 0.8379002213478088, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.37202149629592896, |
| "rewards/length_penalty/std": 0.36219653487205505, |
| "sampling/importance_sampling_ratio/max": 1.440017580986023, |
| "sampling/importance_sampling_ratio/mean": 0.9878596663475037, |
| "sampling/importance_sampling_ratio/min": 0.5892789363861084, |
| "sampling/sampling_logp_difference/max": 0.5288556814193726, |
| "sampling/sampling_logp_difference/mean": 0.020426517352461815, |
| "step": 13, |
| "step_time": 23.57238490600139 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006569349090568722, |
| "clip_ratio/high_mean": 0.0006569349090568722, |
| "clip_ratio/low_mean": 6.0624431353062394e-05, |
| "clip_ratio/low_min": 6.0624431353062394e-05, |
| "clip_ratio/region_mean": 0.0007175593404099345, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 497.0, |
| "completions/max_terminated_length": 497.0, |
| "completions/mean_length": 344.0799865722656, |
| "completions/mean_terminated_length": 344.0799865722656, |
| "completions/min_length": 161.0, |
| "completions/min_terminated_length": 161.0, |
| "entropy": 0.15504273474216462, |
| "epoch": 0.05223880597014925, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00864382367581129, |
| "learning_rate": 5e-05, |
| "loss": -0.0008, |
| "num_tokens": 362682.0, |
| "reward": 0.8119921684265137, |
| "reward_std": 0.14358630776405334, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.16800780594348907, |
| "rewards/length_penalty/std": 0.0466170534491539, |
| "sampling/importance_sampling_ratio/max": 1.39894700050354, |
| "sampling/importance_sampling_ratio/mean": 0.9946113228797913, |
| "sampling/importance_sampling_ratio/min": 0.6841648817062378, |
| "sampling/sampling_logp_difference/max": 0.3795562982559204, |
| "sampling/sampling_logp_difference/mean": 0.011043228209018707, |
| "step": 14, |
| "step_time": 6.20453474111855 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007670841354411095, |
| "clip_ratio/high_mean": 0.0007670841354411095, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007670841354411095, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 706.0, |
| "completions/max_terminated_length": 706.0, |
| "completions/mean_length": 319.6600036621094, |
| "completions/mean_terminated_length": 319.6600036621094, |
| "completions/min_length": 124.0, |
| "completions/min_terminated_length": 124.0, |
| "entropy": 0.160858416557312, |
| "epoch": 0.055970149253731345, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005342600867152214, |
| "learning_rate": 5e-05, |
| "loss": 0.003, |
| "num_tokens": 381195.0, |
| "reward": 0.8039159774780273, |
| "reward_std": 0.19564609229564667, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.15608398616313934, |
| "rewards/length_penalty/std": 0.05970245599746704, |
| "sampling/importance_sampling_ratio/max": 1.4084545373916626, |
| "sampling/importance_sampling_ratio/mean": 0.9941356182098389, |
| "sampling/importance_sampling_ratio/min": 0.7006469368934631, |
| "sampling/sampling_logp_difference/max": 0.3557511568069458, |
| "sampling/sampling_logp_difference/mean": 0.010944732464849949, |
| "step": 15, |
| "step_time": 7.975729608908296 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006197210110258311, |
| "clip_ratio/high_mean": 0.0006197210110258311, |
| "clip_ratio/low_mean": 2.143622696166858e-05, |
| "clip_ratio/low_min": 2.143622696166858e-05, |
| "clip_ratio/region_mean": 0.0006411572394426912, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1765.0, |
| "completions/mean_length": 642.5, |
| "completions/mean_terminated_length": 450.8409118652344, |
| "completions/min_length": 178.0, |
| "completions/min_terminated_length": 178.0, |
| "entropy": 0.16123712956905364, |
| "epoch": 0.05970149253731343, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010248241946101189, |
| "learning_rate": 5e-05, |
| "loss": 0.021, |
| "num_tokens": 416820.0, |
| "reward": 0.5462793111801147, |
| "reward_std": 0.5952073335647583, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.313720703125, |
| "rewards/length_penalty/std": 0.3165541887283325, |
| "sampling/importance_sampling_ratio/max": 1.542428970336914, |
| "sampling/importance_sampling_ratio/mean": 0.993705689907074, |
| "sampling/importance_sampling_ratio/min": 0.666167140007019, |
| "sampling/sampling_logp_difference/max": 0.43335843086242676, |
| "sampling/sampling_logp_difference/mean": 0.011193514801561832, |
| "step": 16, |
| "step_time": 22.68088135495782 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047770159144420175, |
| "clip_ratio/high_mean": 0.00047770159144420175, |
| "clip_ratio/low_mean": 4.515182226896286e-05, |
| "clip_ratio/low_min": 4.515182226896286e-05, |
| "clip_ratio/region_mean": 0.0005228534137131646, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1655.0, |
| "completions/mean_length": 748.8399658203125, |
| "completions/mean_terminated_length": 665.9148559570312, |
| "completions/min_length": 153.0, |
| "completions/min_terminated_length": 153.0, |
| "entropy": 0.1844419479370117, |
| "epoch": 0.06343283582089553, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010951795615255833, |
| "learning_rate": 5e-05, |
| "loss": 0.0253, |
| "num_tokens": 458552.0, |
| "reward": 0.29435545206069946, |
| "reward_std": 0.6421135663986206, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.36564454436302185, |
| "rewards/length_penalty/std": 0.24128037691116333, |
| "sampling/importance_sampling_ratio/max": 1.4502041339874268, |
| "sampling/importance_sampling_ratio/mean": 0.9930887222290039, |
| "sampling/importance_sampling_ratio/min": 0.5193933844566345, |
| "sampling/sampling_logp_difference/max": 0.6550936698913574, |
| "sampling/sampling_logp_difference/mean": 0.012311195023357868, |
| "step": 17, |
| "step_time": 23.103129640687257 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004513544321525842, |
| "clip_ratio/high_mean": 0.0004513544321525842, |
| "clip_ratio/low_mean": 6.295247003436088e-05, |
| "clip_ratio/low_min": 6.295247003436088e-05, |
| "clip_ratio/region_mean": 0.0005143069021869451, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 677.0, |
| "completions/max_terminated_length": 677.0, |
| "completions/mean_length": 314.3399963378906, |
| "completions/mean_terminated_length": 314.3399963378906, |
| "completions/min_length": 160.0, |
| "completions/min_terminated_length": 160.0, |
| "entropy": 0.17831953316926957, |
| "epoch": 0.06716417910447761, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010550854727625847, |
| "learning_rate": 5e-05, |
| "loss": -0.0079, |
| "num_tokens": 477099.0, |
| "reward": 0.8265136480331421, |
| "reward_std": 0.14243507385253906, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.15348632633686066, |
| "rewards/length_penalty/std": 0.043690312653779984, |
| "sampling/importance_sampling_ratio/max": 1.4442710876464844, |
| "sampling/importance_sampling_ratio/mean": 0.9933614730834961, |
| "sampling/importance_sampling_ratio/min": 0.655369758605957, |
| "sampling/sampling_logp_difference/max": 0.42255568504333496, |
| "sampling/sampling_logp_difference/mean": 0.012269136495888233, |
| "step": 18, |
| "step_time": 8.20278396178037 |
| }, |
| { |
| "clip_ratio/high_max": 0.00040143082151189446, |
| "clip_ratio/high_mean": 0.00040143082151189446, |
| "clip_ratio/low_mean": 4.793863918166608e-05, |
| "clip_ratio/low_min": 4.793863918166608e-05, |
| "clip_ratio/region_mean": 0.0004493694577831775, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 993.0, |
| "completions/max_terminated_length": 993.0, |
| "completions/mean_length": 428.5799865722656, |
| "completions/mean_terminated_length": 428.5799865722656, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.14073365926742554, |
| "epoch": 0.0708955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007828718051314354, |
| "learning_rate": 5e-05, |
| "loss": 0.0401, |
| "num_tokens": 501738.0, |
| "reward": 0.7907323837280273, |
| "reward_std": 0.11263526231050491, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.20926757156848907, |
| "rewards/length_penalty/std": 0.11263526231050491, |
| "sampling/importance_sampling_ratio/max": 1.4394450187683105, |
| "sampling/importance_sampling_ratio/mean": 0.9950299263000488, |
| "sampling/importance_sampling_ratio/min": 0.6724317669868469, |
| "sampling/sampling_logp_difference/max": 0.3968546390533447, |
| "sampling/sampling_logp_difference/mean": 0.00956618320196867, |
| "step": 19, |
| "step_time": 11.16728618182242 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005741502449382096, |
| "clip_ratio/high_mean": 0.0005741502449382096, |
| "clip_ratio/low_mean": 0.00018747432623058557, |
| "clip_ratio/low_min": 0.00018747432623058557, |
| "clip_ratio/region_mean": 0.0007616245828103274, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1903.0, |
| "completions/mean_length": 576.6599731445312, |
| "completions/mean_terminated_length": 515.3541870117188, |
| "completions/min_length": 119.0, |
| "completions/min_terminated_length": 119.0, |
| "entropy": 0.21973758041858674, |
| "epoch": 0.07462686567164178, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013793938793241978, |
| "learning_rate": 5e-05, |
| "loss": 0.0493, |
| "num_tokens": 533481.0, |
| "reward": 0.5584277510643005, |
| "reward_std": 0.5993396639823914, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.28157225251197815, |
| "rewards/length_penalty/std": 0.26109856367111206, |
| "sampling/importance_sampling_ratio/max": 1.4702125787734985, |
| "sampling/importance_sampling_ratio/mean": 0.9921209812164307, |
| "sampling/importance_sampling_ratio/min": 0.6277052164077759, |
| "sampling/sampling_logp_difference/max": 0.4656846523284912, |
| "sampling/sampling_logp_difference/mean": 0.014733879826962948, |
| "step": 20, |
| "step_time": 22.10438237595372 |
| }, |
| { |
| "clip_ratio/high_max": 0.00031086085364222525, |
| "clip_ratio/high_mean": 0.00031086085364222525, |
| "clip_ratio/low_mean": 0.0001021033269353211, |
| "clip_ratio/low_min": 0.0001021033269353211, |
| "clip_ratio/region_mean": 0.00041296418057754634, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1112.0, |
| "completions/mean_length": 561.2000122070312, |
| "completions/mean_terminated_length": 319.16278076171875, |
| "completions/min_length": 189.0, |
| "completions/min_terminated_length": 189.0, |
| "entropy": 0.2702585756778717, |
| "epoch": 0.07835820895522388, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009658691473305225, |
| "learning_rate": 5e-05, |
| "loss": 0.0609, |
| "num_tokens": 564361.0, |
| "reward": 0.1259765625, |
| "reward_std": 0.6733714938163757, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.2740234434604645, |
| "rewards/length_penalty/std": 0.3115502595901489, |
| "sampling/importance_sampling_ratio/max": 1.3776147365570068, |
| "sampling/importance_sampling_ratio/mean": 0.9904045462608337, |
| "sampling/importance_sampling_ratio/min": 0.6565403342247009, |
| "sampling/sampling_logp_difference/max": 0.42077112197875977, |
| "sampling/sampling_logp_difference/mean": 0.01713278517127037, |
| "step": 21, |
| "step_time": 22.27446843800135 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004339196544606239, |
| "clip_ratio/high_mean": 0.0004339196544606239, |
| "clip_ratio/low_mean": 8.302292844746262e-05, |
| "clip_ratio/low_min": 8.302292844746262e-05, |
| "clip_ratio/region_mean": 0.0005169425858184695, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1624.0, |
| "completions/max_terminated_length": 1624.0, |
| "completions/mean_length": 403.7599792480469, |
| "completions/mean_terminated_length": 403.7599792480469, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.19005905389785765, |
| "epoch": 0.08208955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007419153116643429, |
| "learning_rate": 5e-05, |
| "loss": 0.0107, |
| "num_tokens": 587929.0, |
| "reward": 0.622851550579071, |
| "reward_std": 0.41565608978271484, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.19714844226837158, |
| "rewards/length_penalty/std": 0.18109971284866333, |
| "sampling/importance_sampling_ratio/max": 1.5093271732330322, |
| "sampling/importance_sampling_ratio/mean": 0.9929693937301636, |
| "sampling/importance_sampling_ratio/min": 0.5926903486251831, |
| "sampling/sampling_logp_difference/max": 0.5230832099914551, |
| "sampling/sampling_logp_difference/mean": 0.01249151211231947, |
| "step": 22, |
| "step_time": 17.272152825258672 |
| }, |
| { |
| "clip_ratio/high_max": 0.00048263907665386797, |
| "clip_ratio/high_mean": 0.00048263907665386797, |
| "clip_ratio/low_mean": 0.00017840272339526565, |
| "clip_ratio/low_min": 0.00017840272339526565, |
| "clip_ratio/region_mean": 0.0006610418146010488, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1995.0, |
| "completions/mean_length": 783.2799682617188, |
| "completions/mean_terminated_length": 673.3043823242188, |
| "completions/min_length": 209.0, |
| "completions/min_terminated_length": 209.0, |
| "entropy": 0.1841371864080429, |
| "epoch": 0.08582089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016199175268411636, |
| "learning_rate": 5e-05, |
| "loss": 0.0047, |
| "num_tokens": 630063.0, |
| "reward": 0.21753905713558197, |
| "reward_std": 0.6258786916732788, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.38246095180511475, |
| "rewards/length_penalty/std": 0.2692579925060272, |
| "sampling/importance_sampling_ratio/max": 1.5615938901901245, |
| "sampling/importance_sampling_ratio/mean": 0.993371307849884, |
| "sampling/importance_sampling_ratio/min": 0.6512693166732788, |
| "sampling/sampling_logp_difference/max": 0.4457070827484131, |
| "sampling/sampling_logp_difference/mean": 0.012494664639234543, |
| "step": 23, |
| "step_time": 23.557205772260204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005802198720630258, |
| "clip_ratio/high_mean": 0.0005802198720630258, |
| "clip_ratio/low_mean": 8.077544625848532e-05, |
| "clip_ratio/low_min": 8.077544625848532e-05, |
| "clip_ratio/region_mean": 0.0006609953183215111, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 902.0, |
| "completions/max_terminated_length": 902.0, |
| "completions/mean_length": 277.7599792480469, |
| "completions/mean_terminated_length": 277.7599792480469, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.21218936145305634, |
| "epoch": 0.08955223880597014, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008748773485422134, |
| "learning_rate": 5e-05, |
| "loss": -0.006, |
| "num_tokens": 646491.0, |
| "reward": 0.6043750047683716, |
| "reward_std": 0.4538353979587555, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.13562500476837158, |
| "rewards/length_penalty/std": 0.06895527988672256, |
| "sampling/importance_sampling_ratio/max": 1.4516805410385132, |
| "sampling/importance_sampling_ratio/mean": 0.992822527885437, |
| "sampling/importance_sampling_ratio/min": 0.46982115507125854, |
| "sampling/sampling_logp_difference/max": 0.7554031610488892, |
| "sampling/sampling_logp_difference/mean": 0.014432352036237717, |
| "step": 24, |
| "step_time": 9.59037149976939 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012464232742786408, |
| "clip_ratio/high_mean": 0.0012464232742786408, |
| "clip_ratio/low_mean": 7.895775488577783e-05, |
| "clip_ratio/low_min": 7.895775488577783e-05, |
| "clip_ratio/region_mean": 0.0013253810349851847, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 335.0, |
| "completions/max_terminated_length": 335.0, |
| "completions/mean_length": 219.1999969482422, |
| "completions/mean_terminated_length": 219.1999969482422, |
| "completions/min_length": 68.0, |
| "completions/min_terminated_length": 68.0, |
| "entropy": 0.17119805812835692, |
| "epoch": 0.09328358208955224, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007603586185723543, |
| "learning_rate": 5e-05, |
| "loss": 0.0056, |
| "num_tokens": 659601.0, |
| "reward": 0.8329687118530273, |
| "reward_std": 0.2540668249130249, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.10703124850988388, |
| "rewards/length_penalty/std": 0.038084786385297775, |
| "sampling/importance_sampling_ratio/max": 1.4122538566589355, |
| "sampling/importance_sampling_ratio/mean": 0.993952214717865, |
| "sampling/importance_sampling_ratio/min": 0.6891739368438721, |
| "sampling/sampling_logp_difference/max": 0.372261643409729, |
| "sampling/sampling_logp_difference/mean": 0.012025420553982258, |
| "step": 25, |
| "step_time": 4.525738164084032 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047754954139236363, |
| "clip_ratio/high_mean": 0.00047754954139236363, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00047754954139236363, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 750.0, |
| "completions/max_terminated_length": 750.0, |
| "completions/mean_length": 419.0, |
| "completions/mean_terminated_length": 419.0, |
| "completions/min_length": 133.0, |
| "completions/min_terminated_length": 133.0, |
| "entropy": 0.1435110628604889, |
| "epoch": 0.09701492537313433, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009241820313036442, |
| "learning_rate": 5e-05, |
| "loss": 0.0221, |
| "num_tokens": 683551.0, |
| "reward": 0.5354101657867432, |
| "reward_std": 0.49858328700065613, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.20458984375, |
| "rewards/length_penalty/std": 0.075383760035038, |
| "sampling/importance_sampling_ratio/max": 1.4293543100357056, |
| "sampling/importance_sampling_ratio/mean": 0.9951713681221008, |
| "sampling/importance_sampling_ratio/min": 0.672234296798706, |
| "sampling/sampling_logp_difference/max": 0.39714837074279785, |
| "sampling/sampling_logp_difference/mean": 0.009974642656743526, |
| "step": 26, |
| "step_time": 9.003170667681843 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008857163833454251, |
| "clip_ratio/high_mean": 0.0008857163833454251, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008857163833454251, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1053.0, |
| "completions/max_terminated_length": 1053.0, |
| "completions/mean_length": 459.47998046875, |
| "completions/mean_terminated_length": 459.47998046875, |
| "completions/min_length": 67.0, |
| "completions/min_terminated_length": 67.0, |
| "entropy": 0.12435056418180465, |
| "epoch": 0.10074626865671642, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012360827066004276, |
| "learning_rate": 5e-05, |
| "loss": 0.0186, |
| "num_tokens": 709325.0, |
| "reward": 0.735644519329071, |
| "reward_std": 0.26206591725349426, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.22435547411441803, |
| "rewards/length_penalty/std": 0.11961404979228973, |
| "sampling/importance_sampling_ratio/max": 1.75804603099823, |
| "sampling/importance_sampling_ratio/mean": 0.9954752922058105, |
| "sampling/importance_sampling_ratio/min": 0.6555108428001404, |
| "sampling/sampling_logp_difference/max": 0.5642030239105225, |
| "sampling/sampling_logp_difference/mean": 0.008994882926344872, |
| "step": 27, |
| "step_time": 11.897754887817428 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047460634959861636, |
| "clip_ratio/high_mean": 0.00047460634959861636, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00047460634959861636, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 802.0, |
| "completions/max_terminated_length": 802.0, |
| "completions/mean_length": 355.6199951171875, |
| "completions/mean_terminated_length": 355.6199951171875, |
| "completions/min_length": 162.0, |
| "completions/min_terminated_length": 162.0, |
| "entropy": 0.15100494623184205, |
| "epoch": 0.1044776119402985, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007981544360518456, |
| "learning_rate": 5e-05, |
| "loss": -0.0147, |
| "num_tokens": 729696.0, |
| "reward": 0.8063573837280273, |
| "reward_std": 0.16736505925655365, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1736425757408142, |
| "rewards/length_penalty/std": 0.08274995535612106, |
| "sampling/importance_sampling_ratio/max": 1.3540838956832886, |
| "sampling/importance_sampling_ratio/mean": 0.9946788549423218, |
| "sampling/importance_sampling_ratio/min": 0.6979334354400635, |
| "sampling/sampling_logp_difference/max": 0.3596315383911133, |
| "sampling/sampling_logp_difference/mean": 0.01037849672138691, |
| "step": 28, |
| "step_time": 9.409750779857859 |
| }, |
| { |
| "clip_ratio/high_max": 0.00023871174198575318, |
| "clip_ratio/high_mean": 0.00023871174198575318, |
| "clip_ratio/low_mean": 0.00039373921463266016, |
| "clip_ratio/low_min": 0.00039373921463266016, |
| "clip_ratio/region_mean": 0.0006324509595287964, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 843.0, |
| "completions/mean_length": 617.3399658203125, |
| "completions/mean_terminated_length": 303.29266357421875, |
| "completions/min_length": 140.0, |
| "completions/min_terminated_length": 140.0, |
| "entropy": 0.34604419469833375, |
| "epoch": 0.10820895522388059, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013593710958957672, |
| "learning_rate": 5e-05, |
| "loss": 0.0773, |
| "num_tokens": 763683.0, |
| "reward": 0.5185644626617432, |
| "reward_std": 0.7207738757133484, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.30143555998802185, |
| "rewards/length_penalty/std": 0.3350928723812103, |
| "sampling/importance_sampling_ratio/max": 1.4225714206695557, |
| "sampling/importance_sampling_ratio/mean": 0.9873183965682983, |
| "sampling/importance_sampling_ratio/min": 0.6167547702789307, |
| "sampling/sampling_logp_difference/max": 0.48328375816345215, |
| "sampling/sampling_logp_difference/mean": 0.021153749898076057, |
| "step": 29, |
| "step_time": 22.45293648680672 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004878065432421863, |
| "clip_ratio/high_mean": 0.0004878065432421863, |
| "clip_ratio/low_mean": 9.996298467740417e-05, |
| "clip_ratio/low_min": 9.996298467740417e-05, |
| "clip_ratio/region_mean": 0.0005877695279195905, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 670.0, |
| "completions/max_terminated_length": 670.0, |
| "completions/mean_length": 357.6600036621094, |
| "completions/mean_terminated_length": 357.6600036621094, |
| "completions/min_length": 205.0, |
| "completions/min_terminated_length": 205.0, |
| "entropy": 0.12727907598018645, |
| "epoch": 0.11194029850746269, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008195790462195873, |
| "learning_rate": 5e-05, |
| "loss": 0.0067, |
| "num_tokens": 784466.0, |
| "reward": 0.6453613042831421, |
| "reward_std": 0.4348297119140625, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.17463867366313934, |
| "rewards/length_penalty/std": 0.05574840307235718, |
| "sampling/importance_sampling_ratio/max": 1.4086308479309082, |
| "sampling/importance_sampling_ratio/mean": 0.9955052137374878, |
| "sampling/importance_sampling_ratio/min": 0.6631681323051453, |
| "sampling/sampling_logp_difference/max": 0.41072678565979004, |
| "sampling/sampling_logp_difference/mean": 0.009073731489479542, |
| "step": 30, |
| "step_time": 7.846059761708602 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008346163784153759, |
| "clip_ratio/high_mean": 0.0008346163784153759, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008346163784153759, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 901.0, |
| "completions/max_terminated_length": 901.0, |
| "completions/mean_length": 390.41998291015625, |
| "completions/mean_terminated_length": 390.41998291015625, |
| "completions/min_length": 232.0, |
| "completions/min_terminated_length": 232.0, |
| "entropy": 0.14007879197597503, |
| "epoch": 0.11567164179104478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009134380146861076, |
| "learning_rate": 5e-05, |
| "loss": -0.0082, |
| "num_tokens": 806957.0, |
| "reward": 0.7493652105331421, |
| "reward_std": 0.24023644626140594, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.19063477218151093, |
| "rewards/length_penalty/std": 0.09261806309223175, |
| "sampling/importance_sampling_ratio/max": 1.8331904411315918, |
| "sampling/importance_sampling_ratio/mean": 0.9952817559242249, |
| "sampling/importance_sampling_ratio/min": 0.6846829056739807, |
| "sampling/sampling_logp_difference/max": 0.60605788230896, |
| "sampling/sampling_logp_difference/mean": 0.00987161509692669, |
| "step": 31, |
| "step_time": 10.142994680441916 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005887864099349827, |
| "clip_ratio/high_mean": 0.0005887864099349827, |
| "clip_ratio/low_mean": 0.00011826351401396095, |
| "clip_ratio/low_min": 0.00011826351401396095, |
| "clip_ratio/region_mean": 0.0007070499297697097, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 491.0, |
| "completions/max_terminated_length": 491.0, |
| "completions/mean_length": 348.6000061035156, |
| "completions/mean_terminated_length": 348.6000061035156, |
| "completions/min_length": 184.0, |
| "completions/min_terminated_length": 184.0, |
| "entropy": 0.16198575794696807, |
| "epoch": 0.11940298507462686, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00773199088871479, |
| "learning_rate": 5e-05, |
| "loss": 0.02, |
| "num_tokens": 826847.0, |
| "reward": 0.6297851204872131, |
| "reward_std": 0.4264633357524872, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.17021484673023224, |
| "rewards/length_penalty/std": 0.04521014913916588, |
| "sampling/importance_sampling_ratio/max": 1.4569092988967896, |
| "sampling/importance_sampling_ratio/mean": 0.9945359826087952, |
| "sampling/importance_sampling_ratio/min": 0.6699723601341248, |
| "sampling/sampling_logp_difference/max": 0.40051889419555664, |
| "sampling/sampling_logp_difference/mean": 0.011234425939619541, |
| "step": 32, |
| "step_time": 6.213606116129085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006170272710733116, |
| "clip_ratio/high_mean": 0.0006170272710733116, |
| "clip_ratio/low_mean": 0.0001180622261017561, |
| "clip_ratio/low_min": 0.0001180622261017561, |
| "clip_ratio/region_mean": 0.0007350894971750677, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1702.0, |
| "completions/max_terminated_length": 1702.0, |
| "completions/mean_length": 604.6599731445312, |
| "completions/mean_terminated_length": 604.6599731445312, |
| "completions/min_length": 153.0, |
| "completions/min_terminated_length": 153.0, |
| "entropy": 0.21591187119483948, |
| "epoch": 0.12313432835820895, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01028301753103733, |
| "learning_rate": 5e-05, |
| "loss": 0.055, |
| "num_tokens": 861810.0, |
| "reward": 0.5047558546066284, |
| "reward_std": 0.5704046487808228, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.29524412751197815, |
| "rewards/length_penalty/std": 0.19361507892608643, |
| "sampling/importance_sampling_ratio/max": 1.4207671880722046, |
| "sampling/importance_sampling_ratio/mean": 0.9924890995025635, |
| "sampling/importance_sampling_ratio/min": 0.5360985994338989, |
| "sampling/sampling_logp_difference/max": 0.6234371662139893, |
| "sampling/sampling_logp_difference/mean": 0.014163421466946602, |
| "step": 33, |
| "step_time": 19.785141061991453 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007411100086756051, |
| "clip_ratio/high_mean": 0.0007411100086756051, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007411100086756051, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 743.0, |
| "completions/max_terminated_length": 743.0, |
| "completions/mean_length": 282.1999816894531, |
| "completions/mean_terminated_length": 282.1999816894531, |
| "completions/min_length": 75.0, |
| "completions/min_terminated_length": 75.0, |
| "entropy": 0.14076103866100312, |
| "epoch": 0.12686567164179105, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0073559340089559555, |
| "learning_rate": 5e-05, |
| "loss": 0.004, |
| "num_tokens": 878860.0, |
| "reward": 0.8222070336341858, |
| "reward_std": 0.244967982172966, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.13779297471046448, |
| "rewards/length_penalty/std": 0.09619967639446259, |
| "sampling/importance_sampling_ratio/max": 1.4278913736343384, |
| "sampling/importance_sampling_ratio/mean": 0.9952441453933716, |
| "sampling/importance_sampling_ratio/min": 0.6817206144332886, |
| "sampling/sampling_logp_difference/max": 0.3831353187561035, |
| "sampling/sampling_logp_difference/mean": 0.009804938919842243, |
| "step": 34, |
| "step_time": 8.534680143930018 |
| }, |
| { |
| "clip_ratio/high_max": 0.001109963731141761, |
| "clip_ratio/high_mean": 0.001109963731141761, |
| "clip_ratio/low_mean": 0.0001634057145565748, |
| "clip_ratio/low_min": 0.0001634057145565748, |
| "clip_ratio/region_mean": 0.0012733694398775696, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1308.0, |
| "completions/mean_length": 495.94000244140625, |
| "completions/mean_terminated_length": 464.2652893066406, |
| "completions/min_length": 237.0, |
| "completions/min_terminated_length": 237.0, |
| "entropy": 0.1782270073890686, |
| "epoch": 0.13059701492537312, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009565229527652264, |
| "learning_rate": 5e-05, |
| "loss": 0.0662, |
| "num_tokens": 906687.0, |
| "reward": 0.5578417778015137, |
| "reward_std": 0.5076192021369934, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.2421582043170929, |
| "rewards/length_penalty/std": 0.14971087872982025, |
| "sampling/importance_sampling_ratio/max": 1.4500327110290527, |
| "sampling/importance_sampling_ratio/mean": 0.9938687086105347, |
| "sampling/importance_sampling_ratio/min": 0.6587642431259155, |
| "sampling/sampling_logp_difference/max": 0.4173896312713623, |
| "sampling/sampling_logp_difference/mean": 0.011907660402357578, |
| "step": 35, |
| "step_time": 21.207997231045738 |
| }, |
| { |
| "clip_ratio/high_max": 0.000516164698638022, |
| "clip_ratio/high_mean": 0.000516164698638022, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000516164698638022, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 509.0, |
| "completions/max_terminated_length": 509.0, |
| "completions/mean_length": 272.5799865722656, |
| "completions/mean_terminated_length": 272.5799865722656, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.14878032505512237, |
| "epoch": 0.13432835820895522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0072594922967255116, |
| "learning_rate": 5e-05, |
| "loss": 0.0178, |
| "num_tokens": 924236.0, |
| "reward": 0.8669042587280273, |
| "reward_std": 0.04621242731809616, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.13309569656848907, |
| "rewards/length_penalty/std": 0.04621243476867676, |
| "sampling/importance_sampling_ratio/max": 1.2858526706695557, |
| "sampling/importance_sampling_ratio/mean": 0.9944921731948853, |
| "sampling/importance_sampling_ratio/min": 0.6621037721633911, |
| "sampling/sampling_logp_difference/max": 0.41233301162719727, |
| "sampling/sampling_logp_difference/mean": 0.010732145980000496, |
| "step": 36, |
| "step_time": 6.420223196735606 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009299026743974537, |
| "clip_ratio/high_mean": 0.0009299026743974537, |
| "clip_ratio/low_mean": 2.9515937785618008e-05, |
| "clip_ratio/low_min": 2.9515937785618008e-05, |
| "clip_ratio/region_mean": 0.0009594186092726886, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1539.0, |
| "completions/mean_length": 572.1599731445312, |
| "completions/mean_terminated_length": 542.0408325195312, |
| "completions/min_length": 213.0, |
| "completions/min_terminated_length": 213.0, |
| "entropy": 0.1613151341676712, |
| "epoch": 0.13805970149253732, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020313387736678123, |
| "learning_rate": 5e-05, |
| "loss": 0.0605, |
| "num_tokens": 956624.0, |
| "reward": 0.5006250143051147, |
| "reward_std": 0.4803377687931061, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.27937498688697815, |
| "rewards/length_penalty/std": 0.18896456062793732, |
| "sampling/importance_sampling_ratio/max": 1.3775994777679443, |
| "sampling/importance_sampling_ratio/mean": 0.9941367506980896, |
| "sampling/importance_sampling_ratio/min": 0.5944056510925293, |
| "sampling/sampling_logp_difference/max": 0.520193338394165, |
| "sampling/sampling_logp_difference/mean": 0.011258355341851711, |
| "step": 37, |
| "step_time": 22.149618682917207 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007627646613400429, |
| "clip_ratio/high_mean": 0.0007627646613400429, |
| "clip_ratio/low_mean": 5.824111867696047e-05, |
| "clip_ratio/low_min": 5.824111867696047e-05, |
| "clip_ratio/region_mean": 0.0008210057800170034, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1161.0, |
| "completions/max_terminated_length": 1161.0, |
| "completions/mean_length": 346.47998046875, |
| "completions/mean_terminated_length": 346.47998046875, |
| "completions/min_length": 101.0, |
| "completions/min_terminated_length": 101.0, |
| "entropy": 0.16285331845283507, |
| "epoch": 0.1417910447761194, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006596951279789209, |
| "learning_rate": 5e-05, |
| "loss": -0.0032, |
| "num_tokens": 976898.0, |
| "reward": 0.790820300579071, |
| "reward_std": 0.21383918821811676, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.16917969286441803, |
| "rewards/length_penalty/std": 0.08555693179368973, |
| "sampling/importance_sampling_ratio/max": 1.4403197765350342, |
| "sampling/importance_sampling_ratio/mean": 0.9942667484283447, |
| "sampling/importance_sampling_ratio/min": 0.6699804663658142, |
| "sampling/sampling_logp_difference/max": 0.40050673484802246, |
| "sampling/sampling_logp_difference/mean": 0.011438596062362194, |
| "step": 38, |
| "step_time": 12.375646460102871 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006980858219321817, |
| "clip_ratio/high_mean": 0.0006980858219321817, |
| "clip_ratio/low_mean": 0.00033374534104950725, |
| "clip_ratio/low_min": 0.00033374534104950725, |
| "clip_ratio/region_mean": 0.0010318311455193908, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 553.0, |
| "completions/max_terminated_length": 553.0, |
| "completions/mean_length": 315.1999816894531, |
| "completions/mean_terminated_length": 315.1999816894531, |
| "completions/min_length": 173.0, |
| "completions/min_terminated_length": 173.0, |
| "entropy": 0.18546728193759918, |
| "epoch": 0.1455223880597015, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008189897052943707, |
| "learning_rate": 5e-05, |
| "loss": -0.0008, |
| "num_tokens": 996928.0, |
| "reward": 0.7660937309265137, |
| "reward_std": 0.275716632604599, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.15390625596046448, |
| "rewards/length_penalty/std": 0.04370494186878204, |
| "sampling/importance_sampling_ratio/max": 1.3966577053070068, |
| "sampling/importance_sampling_ratio/mean": 0.9928783774375916, |
| "sampling/importance_sampling_ratio/min": 0.6465047001838684, |
| "sampling/sampling_logp_difference/max": 0.4361748695373535, |
| "sampling/sampling_logp_difference/mean": 0.012611854821443558, |
| "step": 39, |
| "step_time": 6.961889478377998 |
| }, |
| { |
| "clip_ratio/high_max": 0.000603052054066211, |
| "clip_ratio/high_mean": 0.000603052054066211, |
| "clip_ratio/low_mean": 3.3433633507229385e-05, |
| "clip_ratio/low_min": 3.3433633507229385e-05, |
| "clip_ratio/region_mean": 0.0006364856963045895, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1014.0, |
| "completions/max_terminated_length": 1014.0, |
| "completions/mean_length": 448.91998291015625, |
| "completions/mean_terminated_length": 448.91998291015625, |
| "completions/min_length": 114.0, |
| "completions/min_terminated_length": 114.0, |
| "entropy": 0.11928653568029404, |
| "epoch": 0.14925373134328357, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009980525821447372, |
| "learning_rate": 5e-05, |
| "loss": 0.0063, |
| "num_tokens": 1021754.0, |
| "reward": 0.5408007502555847, |
| "reward_std": 0.4783734977245331, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.21919922530651093, |
| "rewards/length_penalty/std": 0.10957939177751541, |
| "sampling/importance_sampling_ratio/max": 1.4278860092163086, |
| "sampling/importance_sampling_ratio/mean": 0.995747447013855, |
| "sampling/importance_sampling_ratio/min": 0.6380621790885925, |
| "sampling/sampling_logp_difference/max": 0.44931960105895996, |
| "sampling/sampling_logp_difference/mean": 0.008661217987537384, |
| "step": 40, |
| "step_time": 11.311517247231677 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009210823802277446, |
| "clip_ratio/high_mean": 0.0009210823802277446, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0009210823802277446, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 774.0, |
| "completions/max_terminated_length": 774.0, |
| "completions/mean_length": 379.91998291015625, |
| "completions/mean_terminated_length": 379.91998291015625, |
| "completions/min_length": 152.0, |
| "completions/min_terminated_length": 152.0, |
| "entropy": 0.19627552926540376, |
| "epoch": 0.15298507462686567, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007815813645720482, |
| "learning_rate": 5e-05, |
| "loss": 0.0047, |
| "num_tokens": 1045270.0, |
| "reward": 0.5744921565055847, |
| "reward_std": 0.4610985815525055, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.18550781905651093, |
| "rewards/length_penalty/std": 0.06156429648399353, |
| "sampling/importance_sampling_ratio/max": 1.5082297325134277, |
| "sampling/importance_sampling_ratio/mean": 0.9927440881729126, |
| "sampling/importance_sampling_ratio/min": 0.6554858684539795, |
| "sampling/sampling_logp_difference/max": 0.4223785400390625, |
| "sampling/sampling_logp_difference/mean": 0.013935217633843422, |
| "step": 41, |
| "step_time": 9.35883804736659 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003726975875906646, |
| "clip_ratio/high_mean": 0.0003726975875906646, |
| "clip_ratio/low_mean": 6.68225868139416e-05, |
| "clip_ratio/low_min": 6.68225868139416e-05, |
| "clip_ratio/region_mean": 0.0004395201802253723, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 691.0, |
| "completions/max_terminated_length": 691.0, |
| "completions/mean_length": 265.6600036621094, |
| "completions/mean_terminated_length": 265.6600036621094, |
| "completions/min_length": 48.0, |
| "completions/min_terminated_length": 48.0, |
| "entropy": 0.1420754909515381, |
| "epoch": 0.15671641791044777, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006953230127692223, |
| "learning_rate": 5e-05, |
| "loss": 0.0156, |
| "num_tokens": 1061123.0, |
| "reward": 0.8702831864356995, |
| "reward_std": 0.08517740666866302, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.12971679866313934, |
| "rewards/length_penalty/std": 0.08517740666866302, |
| "sampling/importance_sampling_ratio/max": 1.3862687349319458, |
| "sampling/importance_sampling_ratio/mean": 0.9950312972068787, |
| "sampling/importance_sampling_ratio/min": 0.6636943221092224, |
| "sampling/sampling_logp_difference/max": 0.40993356704711914, |
| "sampling/sampling_logp_difference/mean": 0.010521520860493183, |
| "step": 42, |
| "step_time": 7.802435675170273 |
| }, |
| { |
| "clip_ratio/high_max": 0.00078204206074588, |
| "clip_ratio/high_mean": 0.00078204206074588, |
| "clip_ratio/low_mean": 8.100445265881717e-05, |
| "clip_ratio/low_min": 8.100445265881717e-05, |
| "clip_ratio/region_mean": 0.0008630465308669955, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 517.0, |
| "completions/max_terminated_length": 517.0, |
| "completions/mean_length": 264.05999755859375, |
| "completions/mean_terminated_length": 264.05999755859375, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.13952410519123076, |
| "epoch": 0.16044776119402984, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008012687787413597, |
| "learning_rate": 5e-05, |
| "loss": 0.0102, |
| "num_tokens": 1077246.0, |
| "reward": 0.8510644435882568, |
| "reward_std": 0.16554366052150726, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1289355456829071, |
| "rewards/length_penalty/std": 0.04861735180020332, |
| "sampling/importance_sampling_ratio/max": 1.4597078561782837, |
| "sampling/importance_sampling_ratio/mean": 0.9948740601539612, |
| "sampling/importance_sampling_ratio/min": 0.4672364890575409, |
| "sampling/sampling_logp_difference/max": 0.7609198093414307, |
| "sampling/sampling_logp_difference/mean": 0.010898258537054062, |
| "step": 43, |
| "step_time": 6.558947047917172 |
| }, |
| { |
| "clip_ratio/high_max": 0.00034921083715744317, |
| "clip_ratio/high_mean": 0.00034921083715744317, |
| "clip_ratio/low_mean": 4.2983022285625336e-05, |
| "clip_ratio/low_min": 4.2983022285625336e-05, |
| "clip_ratio/region_mean": 0.0003921938652638346, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 787.0, |
| "completions/max_terminated_length": 787.0, |
| "completions/mean_length": 344.53997802734375, |
| "completions/mean_terminated_length": 344.53997802734375, |
| "completions/min_length": 129.0, |
| "completions/min_terminated_length": 129.0, |
| "entropy": 0.13806558549404144, |
| "epoch": 0.16417910447761194, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009876160882413387, |
| "learning_rate": 5e-05, |
| "loss": 0.0006, |
| "num_tokens": 1098703.0, |
| "reward": 0.5717675685882568, |
| "reward_std": 0.4936331808567047, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.16823242604732513, |
| "rewards/length_penalty/std": 0.10850249230861664, |
| "sampling/importance_sampling_ratio/max": 1.4370813369750977, |
| "sampling/importance_sampling_ratio/mean": 0.9950463771820068, |
| "sampling/importance_sampling_ratio/min": 0.6661191582679749, |
| "sampling/sampling_logp_difference/max": 0.40628671646118164, |
| "sampling/sampling_logp_difference/mean": 0.010221735574305058, |
| "step": 44, |
| "step_time": 9.239534565014765 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007109112164471298, |
| "clip_ratio/high_mean": 0.0007109112164471298, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007109112164471298, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 742.0, |
| "completions/max_terminated_length": 742.0, |
| "completions/mean_length": 367.17999267578125, |
| "completions/mean_terminated_length": 367.17999267578125, |
| "completions/min_length": 154.0, |
| "completions/min_terminated_length": 154.0, |
| "entropy": 0.1567250519990921, |
| "epoch": 0.16791044776119404, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008195226080715656, |
| "learning_rate": 5e-05, |
| "loss": -0.0087, |
| "num_tokens": 1120042.0, |
| "reward": 0.8007128834724426, |
| "reward_std": 0.16236166656017303, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1792871057987213, |
| "rewards/length_penalty/std": 0.08163237571716309, |
| "sampling/importance_sampling_ratio/max": 1.4438996315002441, |
| "sampling/importance_sampling_ratio/mean": 0.9942921996116638, |
| "sampling/importance_sampling_ratio/min": 0.5561022162437439, |
| "sampling/sampling_logp_difference/max": 0.5868031978607178, |
| "sampling/sampling_logp_difference/mean": 0.011431382037699223, |
| "step": 45, |
| "step_time": 8.618486961815506 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006855419371277094, |
| "clip_ratio/high_mean": 0.0006855419371277094, |
| "clip_ratio/low_mean": 7.782101165503264e-05, |
| "clip_ratio/low_min": 7.782101165503264e-05, |
| "clip_ratio/region_mean": 0.000763362948782742, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 487.0, |
| "completions/max_terminated_length": 487.0, |
| "completions/mean_length": 249.33999633789062, |
| "completions/mean_terminated_length": 249.33999633789062, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.15599994957447053, |
| "epoch": 0.17164179104477612, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006997830234467983, |
| "learning_rate": 5e-05, |
| "loss": 0.0051, |
| "num_tokens": 1135789.0, |
| "reward": 0.8382519483566284, |
| "reward_std": 0.20857727527618408, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.12174804508686066, |
| "rewards/length_penalty/std": 0.051770467311143875, |
| "sampling/importance_sampling_ratio/max": 1.47560453414917, |
| "sampling/importance_sampling_ratio/mean": 0.9941657781600952, |
| "sampling/importance_sampling_ratio/min": 0.5952705144882202, |
| "sampling/sampling_logp_difference/max": 0.5187393426895142, |
| "sampling/sampling_logp_difference/mean": 0.012247167527675629, |
| "step": 46, |
| "step_time": 5.923294740961865 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007308489381102845, |
| "clip_ratio/high_mean": 0.0007308489381102845, |
| "clip_ratio/low_mean": 0.0002699727250728756, |
| "clip_ratio/low_min": 0.0002699727250728756, |
| "clip_ratio/region_mean": 0.001000821660272777, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1873.0, |
| "completions/mean_length": 577.0399780273438, |
| "completions/mean_terminated_length": 376.4545593261719, |
| "completions/min_length": 144.0, |
| "completions/min_terminated_length": 144.0, |
| "entropy": 0.3190369069576263, |
| "epoch": 0.17537313432835822, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010635782033205032, |
| "learning_rate": 5e-05, |
| "loss": 0.0455, |
| "num_tokens": 1167911.0, |
| "reward": 0.3182421922683716, |
| "reward_std": 0.7228251099586487, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.28175780177116394, |
| "rewards/length_penalty/std": 0.31979885697364807, |
| "sampling/importance_sampling_ratio/max": 2.8393774032592773, |
| "sampling/importance_sampling_ratio/mean": 0.9887529611587524, |
| "sampling/importance_sampling_ratio/min": 0.6558433771133423, |
| "sampling/sampling_logp_difference/max": 1.0435848236083984, |
| "sampling/sampling_logp_difference/mean": 0.01984455995261669, |
| "step": 47, |
| "step_time": 22.49316789279692 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006271806021686643, |
| "clip_ratio/high_mean": 0.0006271806021686643, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0006271806021686643, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 758.0, |
| "completions/max_terminated_length": 758.0, |
| "completions/mean_length": 317.5799865722656, |
| "completions/mean_terminated_length": 317.5799865722656, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.16717901229858398, |
| "epoch": 0.1791044776119403, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007288388442248106, |
| "learning_rate": 5e-05, |
| "loss": -0.0015, |
| "num_tokens": 1185780.0, |
| "reward": 0.5649316310882568, |
| "reward_std": 0.41125473380088806, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.15506835281848907, |
| "rewards/length_penalty/std": 0.09044893085956573, |
| "sampling/importance_sampling_ratio/max": 2.0443358421325684, |
| "sampling/importance_sampling_ratio/mean": 0.9941331148147583, |
| "sampling/importance_sampling_ratio/min": 0.638364851474762, |
| "sampling/sampling_logp_difference/max": 0.7150728702545166, |
| "sampling/sampling_logp_difference/mean": 0.01257918868213892, |
| "step": 48, |
| "step_time": 8.94219338009134 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007135229592677206, |
| "clip_ratio/high_mean": 0.0007135229592677206, |
| "clip_ratio/low_mean": 3.603603690862656e-05, |
| "clip_ratio/low_min": 3.603603690862656e-05, |
| "clip_ratio/region_mean": 0.0007495589961763471, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1032.0, |
| "completions/max_terminated_length": 1032.0, |
| "completions/mean_length": 506.3599853515625, |
| "completions/mean_terminated_length": 506.3599853515625, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.15755688548088073, |
| "epoch": 0.1828358208955224, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009247946552932262, |
| "learning_rate": 5e-05, |
| "loss": 0.0127, |
| "num_tokens": 1213218.0, |
| "reward": 0.57275390625, |
| "reward_std": 0.45781880617141724, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.24724608659744263, |
| "rewards/length_penalty/std": 0.10344348847866058, |
| "sampling/importance_sampling_ratio/max": 1.4508453607559204, |
| "sampling/importance_sampling_ratio/mean": 0.9941213130950928, |
| "sampling/importance_sampling_ratio/min": 0.5819795727729797, |
| "sampling/sampling_logp_difference/max": 0.5413199663162231, |
| "sampling/sampling_logp_difference/mean": 0.01129181683063507, |
| "step": 49, |
| "step_time": 11.669174040900543 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008831843151710927, |
| "clip_ratio/high_mean": 0.0008831843151710927, |
| "clip_ratio/low_mean": 6.391818751581014e-05, |
| "clip_ratio/low_min": 6.391818751581014e-05, |
| "clip_ratio/region_mean": 0.0009471025085076689, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1457.0, |
| "completions/mean_length": 504.5999755859375, |
| "completions/mean_terminated_length": 473.1020202636719, |
| "completions/min_length": 121.0, |
| "completions/min_terminated_length": 121.0, |
| "entropy": 0.2619792610406876, |
| "epoch": 0.1865671641791045, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014896593056619167, |
| "learning_rate": 5e-05, |
| "loss": -0.0465, |
| "num_tokens": 1241348.0, |
| "reward": 0.47361326217651367, |
| "reward_std": 0.52699214220047, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.24638672173023224, |
| "rewards/length_penalty/std": 0.1892664134502411, |
| "sampling/importance_sampling_ratio/max": 2.098832607269287, |
| "sampling/importance_sampling_ratio/mean": 0.9907480478286743, |
| "sampling/importance_sampling_ratio/min": 0.576706051826477, |
| "sampling/sampling_logp_difference/max": 0.7413812875747681, |
| "sampling/sampling_logp_difference/mean": 0.01703432761132717, |
| "step": 50, |
| "step_time": 21.40806119493209 |
| }, |
| { |
| "clip_ratio/high_max": 0.00019636719953268765, |
| "clip_ratio/high_mean": 0.00019636719953268765, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00019636719953268765, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 468.0, |
| "completions/max_terminated_length": 468.0, |
| "completions/mean_length": 237.37998962402344, |
| "completions/mean_terminated_length": 237.37998962402344, |
| "completions/min_length": 106.0, |
| "completions/min_terminated_length": 106.0, |
| "entropy": 0.1493792414665222, |
| "epoch": 0.19029850746268656, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007988722063601017, |
| "learning_rate": 5e-05, |
| "loss": 0.0223, |
| "num_tokens": 1255427.0, |
| "reward": 0.8840917944908142, |
| "reward_std": 0.0365288220345974, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.11590820550918579, |
| "rewards/length_penalty/std": 0.0365288220345974, |
| "sampling/importance_sampling_ratio/max": 2.037508726119995, |
| "sampling/importance_sampling_ratio/mean": 0.9946038722991943, |
| "sampling/importance_sampling_ratio/min": 0.37826406955718994, |
| "sampling/sampling_logp_difference/max": 0.9721627235412598, |
| "sampling/sampling_logp_difference/mean": 0.012720397673547268, |
| "step": 51, |
| "step_time": 5.5065665282309055 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006384580279700458, |
| "clip_ratio/high_mean": 0.0006384580279700458, |
| "clip_ratio/low_mean": 0.00015159418107941748, |
| "clip_ratio/low_min": 0.00015159418107941748, |
| "clip_ratio/region_mean": 0.0007900522090494633, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2032.0, |
| "completions/mean_length": 568.4599609375, |
| "completions/mean_terminated_length": 439.8043518066406, |
| "completions/min_length": 114.0, |
| "completions/min_terminated_length": 114.0, |
| "entropy": 0.315947163105011, |
| "epoch": 0.19402985074626866, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017853155732154846, |
| "learning_rate": 5e-05, |
| "loss": 0.0819, |
| "num_tokens": 1287180.0, |
| "reward": 0.38243162631988525, |
| "reward_std": 0.6984579563140869, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.27756837010383606, |
| "rewards/length_penalty/std": 0.29989153146743774, |
| "sampling/importance_sampling_ratio/max": 1.4278913736343384, |
| "sampling/importance_sampling_ratio/mean": 0.9887783527374268, |
| "sampling/importance_sampling_ratio/min": 0.44529446959495544, |
| "sampling/sampling_logp_difference/max": 0.8090195655822754, |
| "sampling/sampling_logp_difference/mean": 0.020555591210722923, |
| "step": 52, |
| "step_time": 22.151965332916006 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007838805031497031, |
| "clip_ratio/high_mean": 0.0007838805031497031, |
| "clip_ratio/low_mean": 0.00013619677629321812, |
| "clip_ratio/low_min": 0.00013619677629321812, |
| "clip_ratio/region_mean": 0.0009200772561598569, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 479.0, |
| "completions/max_terminated_length": 479.0, |
| "completions/mean_length": 271.739990234375, |
| "completions/mean_terminated_length": 271.739990234375, |
| "completions/min_length": 131.0, |
| "completions/min_terminated_length": 131.0, |
| "entropy": 0.1576307773590088, |
| "epoch": 0.19776119402985073, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009781831875443459, |
| "learning_rate": 5e-05, |
| "loss": -0.0006, |
| "num_tokens": 1303577.0, |
| "reward": 0.7273144125938416, |
| "reward_std": 0.35877934098243713, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.13268554210662842, |
| "rewards/length_penalty/std": 0.03264938294887543, |
| "sampling/importance_sampling_ratio/max": 2.14555287361145, |
| "sampling/importance_sampling_ratio/mean": 0.9947115182876587, |
| "sampling/importance_sampling_ratio/min": 0.563351035118103, |
| "sampling/sampling_logp_difference/max": 0.763397216796875, |
| "sampling/sampling_logp_difference/mean": 0.013018809258937836, |
| "step": 53, |
| "step_time": 6.220042122993618 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009491689561400563, |
| "clip_ratio/high_mean": 0.0009491689561400563, |
| "clip_ratio/low_mean": 4.91038546897471e-05, |
| "clip_ratio/low_min": 4.91038546897471e-05, |
| "clip_ratio/region_mean": 0.0009982727991882713, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 694.0, |
| "completions/max_terminated_length": 694.0, |
| "completions/mean_length": 353.6600036621094, |
| "completions/mean_terminated_length": 353.6600036621094, |
| "completions/min_length": 142.0, |
| "completions/min_terminated_length": 142.0, |
| "entropy": 0.16784085631370543, |
| "epoch": 0.20149253731343283, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009280281141400337, |
| "learning_rate": 5e-05, |
| "loss": 0.0017, |
| "num_tokens": 1323720.0, |
| "reward": 0.26731443405151367, |
| "reward_std": 0.5046936273574829, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.17268554866313934, |
| "rewards/length_penalty/std": 0.059796757996082306, |
| "sampling/importance_sampling_ratio/max": 1.5517566204071045, |
| "sampling/importance_sampling_ratio/mean": 0.9940769076347351, |
| "sampling/importance_sampling_ratio/min": 0.5662176012992859, |
| "sampling/sampling_logp_difference/max": 0.5687768459320068, |
| "sampling/sampling_logp_difference/mean": 0.013525796122848988, |
| "step": 54, |
| "step_time": 8.069038881920278 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004341446969192475, |
| "clip_ratio/high_mean": 0.0004341446969192475, |
| "clip_ratio/low_mean": 3.74531839042902e-05, |
| "clip_ratio/low_min": 3.74531839042902e-05, |
| "clip_ratio/region_mean": 0.00047159788082353773, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 745.0, |
| "completions/mean_length": 354.47998046875, |
| "completions/mean_terminated_length": 319.9183654785156, |
| "completions/min_length": 106.0, |
| "completions/min_terminated_length": 106.0, |
| "entropy": 0.15769305527210237, |
| "epoch": 0.20522388059701493, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01787675730884075, |
| "learning_rate": 5e-05, |
| "loss": 0.0917, |
| "num_tokens": 1344304.0, |
| "reward": 0.6069140434265137, |
| "reward_std": 0.4861266314983368, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.17308594286441803, |
| "rewards/length_penalty/std": 0.1466371715068817, |
| "sampling/importance_sampling_ratio/max": 1.9533584117889404, |
| "sampling/importance_sampling_ratio/mean": 0.9941394925117493, |
| "sampling/importance_sampling_ratio/min": 0.3172576427459717, |
| "sampling/sampling_logp_difference/max": 1.1480411291122437, |
| "sampling/sampling_logp_difference/mean": 0.013562560081481934, |
| "step": 55, |
| "step_time": 20.474934197962284 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004629950679372996, |
| "clip_ratio/high_mean": 0.0004629950679372996, |
| "clip_ratio/low_mean": 4.444938385859132e-05, |
| "clip_ratio/low_min": 4.444938385859132e-05, |
| "clip_ratio/region_mean": 0.000507444451795891, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2019.0, |
| "completions/mean_length": 609.0, |
| "completions/mean_terminated_length": 334.9047546386719, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.2297052264213562, |
| "epoch": 0.208955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012723367661237717, |
| "learning_rate": 5e-05, |
| "loss": 0.0267, |
| "num_tokens": 1377194.0, |
| "reward": 0.46263670921325684, |
| "reward_std": 0.70781409740448, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.29736328125, |
| "rewards/length_penalty/std": 0.3400876820087433, |
| "sampling/importance_sampling_ratio/max": 2.268763780593872, |
| "sampling/importance_sampling_ratio/mean": 0.9911385774612427, |
| "sampling/importance_sampling_ratio/min": 0.4066545367240906, |
| "sampling/sampling_logp_difference/max": 0.8997912406921387, |
| "sampling/sampling_logp_difference/mean": 0.016253819689154625, |
| "step": 56, |
| "step_time": 22.30247456091456 |
| }, |
| { |
| "clip_ratio/high_max": 0.000757315318332985, |
| "clip_ratio/high_mean": 0.000757315318332985, |
| "clip_ratio/low_mean": 5.267316591925919e-05, |
| "clip_ratio/low_min": 5.267316591925919e-05, |
| "clip_ratio/region_mean": 0.0008099884842522442, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1166.0, |
| "completions/max_terminated_length": 1166.0, |
| "completions/mean_length": 430.9599914550781, |
| "completions/mean_terminated_length": 430.9599914550781, |
| "completions/min_length": 232.0, |
| "completions/min_terminated_length": 232.0, |
| "entropy": 0.17580842673778535, |
| "epoch": 0.2126865671641791, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011164488270878792, |
| "learning_rate": 5e-05, |
| "loss": 0.0504, |
| "num_tokens": 1402612.0, |
| "reward": 0.7895702719688416, |
| "reward_std": 0.07827731221914291, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.21042968332767487, |
| "rewards/length_penalty/std": 0.07827731221914291, |
| "sampling/importance_sampling_ratio/max": 2.252070188522339, |
| "sampling/importance_sampling_ratio/mean": 0.9938055872917175, |
| "sampling/importance_sampling_ratio/min": 0.46339696645736694, |
| "sampling/sampling_logp_difference/max": 0.81184983253479, |
| "sampling/sampling_logp_difference/mean": 0.014916970394551754, |
| "step": 57, |
| "step_time": 12.640353741589934 |
| }, |
| { |
| "clip_ratio/high_max": 0.000622326077427715, |
| "clip_ratio/high_mean": 0.000622326077427715, |
| "clip_ratio/low_mean": 8.936550584621727e-05, |
| "clip_ratio/low_min": 8.936550584621727e-05, |
| "clip_ratio/region_mean": 0.0007116915890946985, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 415.0, |
| "completions/max_terminated_length": 415.0, |
| "completions/mean_length": 223.45999145507812, |
| "completions/mean_terminated_length": 223.45999145507812, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.15358326137065886, |
| "epoch": 0.21641791044776118, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00815771147608757, |
| "learning_rate": 5e-05, |
| "loss": -0.0018, |
| "num_tokens": 1417075.0, |
| "reward": 0.5708886384963989, |
| "reward_std": 0.4770105481147766, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.10911133140325546, |
| "rewards/length_penalty/std": 0.02866869606077671, |
| "sampling/importance_sampling_ratio/max": 2.1270813941955566, |
| "sampling/importance_sampling_ratio/mean": 0.9937748312950134, |
| "sampling/importance_sampling_ratio/min": 0.23586927354335785, |
| "sampling/sampling_logp_difference/max": 1.4444775581359863, |
| "sampling/sampling_logp_difference/mean": 0.016142576932907104, |
| "step": 58, |
| "step_time": 5.718622662592679 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004469272796995938, |
| "clip_ratio/high_mean": 0.0004469272796995938, |
| "clip_ratio/low_mean": 0.000222334690624848, |
| "clip_ratio/low_min": 0.000222334690624848, |
| "clip_ratio/region_mean": 0.000669261981965974, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 431.0, |
| "completions/max_terminated_length": 431.0, |
| "completions/mean_length": 255.25999450683594, |
| "completions/mean_terminated_length": 255.25999450683594, |
| "completions/min_length": 135.0, |
| "completions/min_terminated_length": 135.0, |
| "entropy": 0.14214683771133424, |
| "epoch": 0.22014925373134328, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009184117428958416, |
| "learning_rate": 5e-05, |
| "loss": 0.0083, |
| "num_tokens": 1432748.0, |
| "reward": 0.85536128282547, |
| "reward_std": 0.15642878413200378, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.12463866919279099, |
| "rewards/length_penalty/std": 0.031104518100619316, |
| "sampling/importance_sampling_ratio/max": 2.116948366165161, |
| "sampling/importance_sampling_ratio/mean": 0.9938475489616394, |
| "sampling/importance_sampling_ratio/min": 0.36953774094581604, |
| "sampling/sampling_logp_difference/max": 0.9955024719238281, |
| "sampling/sampling_logp_difference/mean": 0.015187020413577557, |
| "step": 59, |
| "step_time": 5.274361951975152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013223550049588084, |
| "clip_ratio/high_mean": 0.0013223550049588084, |
| "clip_ratio/low_mean": 6.949270609766245e-05, |
| "clip_ratio/low_min": 6.949270609766245e-05, |
| "clip_ratio/region_mean": 0.001391847711056471, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 524.0, |
| "completions/max_terminated_length": 524.0, |
| "completions/mean_length": 310.6600036621094, |
| "completions/mean_terminated_length": 310.6600036621094, |
| "completions/min_length": 143.0, |
| "completions/min_terminated_length": 143.0, |
| "entropy": 0.1767831802368164, |
| "epoch": 0.22388059701492538, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01105369534343481, |
| "learning_rate": 5e-05, |
| "loss": -0.0052, |
| "num_tokens": 1450811.0, |
| "reward": 0.5883105397224426, |
| "reward_std": 0.4327555000782013, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.15168945491313934, |
| "rewards/length_penalty/std": 0.05061161890625954, |
| "sampling/importance_sampling_ratio/max": 2.813166856765747, |
| "sampling/importance_sampling_ratio/mean": 0.9937822818756104, |
| "sampling/importance_sampling_ratio/min": 0.30162176489830017, |
| "sampling/sampling_logp_difference/max": 1.1985814571380615, |
| "sampling/sampling_logp_difference/mean": 0.01754101552069187, |
| "step": 60, |
| "step_time": 6.406686214031652 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011369908577762544, |
| "clip_ratio/high_mean": 0.0011369908577762544, |
| "clip_ratio/low_mean": 0.00026866488624364135, |
| "clip_ratio/low_min": 0.00026866488624364135, |
| "clip_ratio/region_mean": 0.0014056557440198958, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2001.0, |
| "completions/mean_length": 548.3999633789062, |
| "completions/mean_terminated_length": 418.0, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.3446581304073334, |
| "epoch": 0.22761194029850745, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006843577139079571, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 1480661.0, |
| "reward": 0.5122265219688416, |
| "reward_std": 0.7297219634056091, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.26777344942092896, |
| "rewards/length_penalty/std": 0.32890328764915466, |
| "sampling/importance_sampling_ratio/max": 2.2495930194854736, |
| "sampling/importance_sampling_ratio/mean": 0.9877831935882568, |
| "sampling/importance_sampling_ratio/min": 0.24449598789215088, |
| "sampling/sampling_logp_difference/max": 1.4085564613342285, |
| "sampling/sampling_logp_difference/mean": 0.022903194651007652, |
| "step": 61, |
| "step_time": 22.07145461300388 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004500309267314151, |
| "clip_ratio/high_mean": 0.0004500309267314151, |
| "clip_ratio/low_mean": 0.00010034791193902493, |
| "clip_ratio/low_min": 0.00010034791193902493, |
| "clip_ratio/region_mean": 0.00055037883867044, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1895.0, |
| "completions/mean_length": 610.3999633789062, |
| "completions/mean_terminated_length": 376.3721008300781, |
| "completions/min_length": 144.0, |
| "completions/min_terminated_length": 144.0, |
| "entropy": 0.2861504554748535, |
| "epoch": 0.23134328358208955, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017049646005034447, |
| "learning_rate": 5e-05, |
| "loss": 0.0175, |
| "num_tokens": 1513861.0, |
| "reward": 0.5219531059265137, |
| "reward_std": 0.7053924202919006, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.29804688692092896, |
| "rewards/length_penalty/std": 0.3303263485431671, |
| "sampling/importance_sampling_ratio/max": 2.409658193588257, |
| "sampling/importance_sampling_ratio/mean": 0.9900499582290649, |
| "sampling/importance_sampling_ratio/min": 0.15424886345863342, |
| "sampling/sampling_logp_difference/max": 1.8691879510879517, |
| "sampling/sampling_logp_difference/mean": 0.02033442258834839, |
| "step": 62, |
| "step_time": 22.226805459707975 |
| }, |
| { |
| "clip_ratio/high_max": 0.00038249637000262735, |
| "clip_ratio/high_mean": 0.00038249637000262735, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00038249637000262735, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 902.0, |
| "completions/max_terminated_length": 902.0, |
| "completions/mean_length": 309.1999816894531, |
| "completions/mean_terminated_length": 309.1999816894531, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.16854563653469085, |
| "epoch": 0.23507462686567165, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01063678041100502, |
| "learning_rate": 5e-05, |
| "loss": 0.0004, |
| "num_tokens": 1532231.0, |
| "reward": 0.6490234136581421, |
| "reward_std": 0.4646871089935303, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.15097656846046448, |
| "rewards/length_penalty/std": 0.09777788817882538, |
| "sampling/importance_sampling_ratio/max": 2.530210018157959, |
| "sampling/importance_sampling_ratio/mean": 0.9931871891021729, |
| "sampling/importance_sampling_ratio/min": 0.5236321091651917, |
| "sampling/sampling_logp_difference/max": 0.9283022880554199, |
| "sampling/sampling_logp_difference/mean": 0.01587487757205963, |
| "step": 63, |
| "step_time": 10.248050187015906 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002450101776048541, |
| "clip_ratio/high_mean": 0.0002450101776048541, |
| "clip_ratio/low_mean": 9.394082007929683e-05, |
| "clip_ratio/low_min": 9.394082007929683e-05, |
| "clip_ratio/region_mean": 0.0003389509976841509, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 636.0, |
| "completions/max_terminated_length": 636.0, |
| "completions/mean_length": 392.739990234375, |
| "completions/mean_terminated_length": 392.739990234375, |
| "completions/min_length": 204.0, |
| "completions/min_terminated_length": 204.0, |
| "entropy": 0.1435816317796707, |
| "epoch": 0.23880597014925373, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007314858492463827, |
| "learning_rate": 5e-05, |
| "loss": 0.0085, |
| "num_tokens": 1556258.0, |
| "reward": 0.7882323861122131, |
| "reward_std": 0.16649143397808075, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.19176757335662842, |
| "rewards/length_penalty/std": 0.05821884796023369, |
| "sampling/importance_sampling_ratio/max": 2.55224871635437, |
| "sampling/importance_sampling_ratio/mean": 0.9948147535324097, |
| "sampling/importance_sampling_ratio/min": 0.3057543635368347, |
| "sampling/sampling_logp_difference/max": 1.1849732398986816, |
| "sampling/sampling_logp_difference/mean": 0.014711213298141956, |
| "step": 64, |
| "step_time": 7.967395526356995 |
| }, |
| { |
| "clip_ratio/high_max": 0.00042739183409139513, |
| "clip_ratio/high_mean": 0.00042739183409139513, |
| "clip_ratio/low_mean": 0.00015360700781457125, |
| "clip_ratio/low_min": 0.00015360700781457125, |
| "clip_ratio/region_mean": 0.0005809988360852003, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 482.0, |
| "completions/max_terminated_length": 482.0, |
| "completions/mean_length": 280.1999816894531, |
| "completions/mean_terminated_length": 280.1999816894531, |
| "completions/min_length": 113.0, |
| "completions/min_terminated_length": 113.0, |
| "entropy": 0.150139519572258, |
| "epoch": 0.24253731343283583, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007787035778164864, |
| "learning_rate": 5e-05, |
| "loss": 0.0058, |
| "num_tokens": 1572738.0, |
| "reward": 0.563183605670929, |
| "reward_std": 0.49956074357032776, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.13681641221046448, |
| "rewards/length_penalty/std": 0.0541207455098629, |
| "sampling/importance_sampling_ratio/max": 2.791949510574341, |
| "sampling/importance_sampling_ratio/mean": 0.9945249557495117, |
| "sampling/importance_sampling_ratio/min": 0.16231341660022736, |
| "sampling/sampling_logp_difference/max": 1.8182260990142822, |
| "sampling/sampling_logp_difference/mean": 0.017655324190855026, |
| "step": 65, |
| "step_time": 5.998067596228793 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010637091821990908, |
| "clip_ratio/high_mean": 0.0010637091821990908, |
| "clip_ratio/low_mean": 0.00011672378168441355, |
| "clip_ratio/low_min": 0.00011672378168441355, |
| "clip_ratio/region_mean": 0.001180432946421206, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 881.0, |
| "completions/max_terminated_length": 881.0, |
| "completions/mean_length": 342.47998046875, |
| "completions/mean_terminated_length": 342.47998046875, |
| "completions/min_length": 173.0, |
| "completions/min_terminated_length": 173.0, |
| "entropy": 0.22419943511486054, |
| "epoch": 0.2462686567164179, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00808474887162447, |
| "learning_rate": 5e-05, |
| "loss": -0.0, |
| "num_tokens": 1593142.0, |
| "reward": 0.7527734041213989, |
| "reward_std": 0.27381327748298645, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.16722656786441803, |
| "rewards/length_penalty/std": 0.08004416525363922, |
| "sampling/importance_sampling_ratio/max": 2.147165298461914, |
| "sampling/importance_sampling_ratio/mean": 0.9921154379844666, |
| "sampling/importance_sampling_ratio/min": 0.3174567222595215, |
| "sampling/sampling_logp_difference/max": 1.147413730621338, |
| "sampling/sampling_logp_difference/mean": 0.020227329805493355, |
| "step": 66, |
| "step_time": 10.123271165182814 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014360137982293963, |
| "clip_ratio/high_mean": 0.0014360137982293963, |
| "clip_ratio/low_mean": 0.0001180289196781814, |
| "clip_ratio/low_min": 0.0001180289196781814, |
| "clip_ratio/region_mean": 0.0015540427062660455, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1050.0, |
| "completions/max_terminated_length": 1050.0, |
| "completions/mean_length": 290.9599914550781, |
| "completions/mean_terminated_length": 290.9599914550781, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.21816076338291168, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011416135355830193, |
| "learning_rate": 5e-05, |
| "loss": 0.0122, |
| "num_tokens": 1610040.0, |
| "reward": 0.5379296541213989, |
| "reward_std": 0.506062924861908, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.14207030832767487, |
| "rewards/length_penalty/std": 0.10983388125896454, |
| "sampling/importance_sampling_ratio/max": 2.0739052295684814, |
| "sampling/importance_sampling_ratio/mean": 0.9912834167480469, |
| "sampling/importance_sampling_ratio/min": 0.1838129460811615, |
| "sampling/sampling_logp_difference/max": 1.6938366889953613, |
| "sampling/sampling_logp_difference/mean": 0.020820939913392067, |
| "step": 67, |
| "step_time": 11.212650744011626 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008424129860941321, |
| "clip_ratio/high_mean": 0.0008424129860941321, |
| "clip_ratio/low_mean": 4.200357070658356e-05, |
| "clip_ratio/low_min": 4.200357070658356e-05, |
| "clip_ratio/region_mean": 0.0008844165568007156, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1909.0, |
| "completions/mean_length": 560.3800048828125, |
| "completions/mean_terminated_length": 277.0238037109375, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.3256783872842789, |
| "epoch": 0.2537313432835821, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021823778748512268, |
| "learning_rate": 5e-05, |
| "loss": 0.0302, |
| "num_tokens": 1641769.0, |
| "reward": 0.5663769245147705, |
| "reward_std": 0.7070843577384949, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.2736230492591858, |
| "rewards/length_penalty/std": 0.3544589579105377, |
| "sampling/importance_sampling_ratio/max": 2.07163405418396, |
| "sampling/importance_sampling_ratio/mean": 0.9875916838645935, |
| "sampling/importance_sampling_ratio/min": 0.19297704100608826, |
| "sampling/sampling_logp_difference/max": 1.64518404006958, |
| "sampling/sampling_logp_difference/mean": 0.02346353977918625, |
| "step": 68, |
| "step_time": 22.53464117506519 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007057235925458372, |
| "clip_ratio/high_mean": 0.0007057235925458372, |
| "clip_ratio/low_mean": 7.66577257309109e-05, |
| "clip_ratio/low_min": 7.66577257309109e-05, |
| "clip_ratio/region_mean": 0.000782381318276748, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 482.0, |
| "completions/max_terminated_length": 482.0, |
| "completions/mean_length": 233.75999450683594, |
| "completions/mean_terminated_length": 233.75999450683594, |
| "completions/min_length": 80.0, |
| "completions/min_terminated_length": 80.0, |
| "entropy": 0.18384387493133544, |
| "epoch": 0.2574626865671642, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006017310544848442, |
| "learning_rate": 5e-05, |
| "loss": -0.0044, |
| "num_tokens": 1655987.0, |
| "reward": 0.8258593678474426, |
| "reward_std": 0.23231732845306396, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.11414062231779099, |
| "rewards/length_penalty/std": 0.05983913317322731, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927144050598145, |
| "sampling/importance_sampling_ratio/min": 0.16751541197299957, |
| "sampling/sampling_logp_difference/max": 1.786679983139038, |
| "sampling/sampling_logp_difference/mean": 0.020872505381703377, |
| "step": 69, |
| "step_time": 6.137518534902483 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015542366134468466, |
| "clip_ratio/high_mean": 0.0015542366134468466, |
| "clip_ratio/low_mean": 0.0001427293347660452, |
| "clip_ratio/low_min": 0.0001427293347660452, |
| "clip_ratio/region_mean": 0.001696965959854424, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 416.0, |
| "completions/mean_length": 229.77999877929688, |
| "completions/mean_terminated_length": 192.6734619140625, |
| "completions/min_length": 56.0, |
| "completions/min_terminated_length": 56.0, |
| "entropy": 0.23980919718742372, |
| "epoch": 0.26119402985074625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02051672153174877, |
| "learning_rate": 5e-05, |
| "loss": 0.1113, |
| "num_tokens": 1669906.0, |
| "reward": 0.8678027391433716, |
| "reward_std": 0.27289706468582153, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.11219726502895355, |
| "rewards/length_penalty/std": 0.13504061102867126, |
| "sampling/importance_sampling_ratio/max": 2.243411064147949, |
| "sampling/importance_sampling_ratio/mean": 0.9905449748039246, |
| "sampling/importance_sampling_ratio/min": 0.24543412029743195, |
| "sampling/sampling_logp_difference/max": 1.4047267436981201, |
| "sampling/sampling_logp_difference/mean": 0.0255883876234293, |
| "step": 70, |
| "step_time": 19.926869072951376 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010858166177058592, |
| "clip_ratio/high_mean": 0.0010858166177058592, |
| "clip_ratio/low_mean": 0.0001210609101690352, |
| "clip_ratio/low_min": 0.0001210609101690352, |
| "clip_ratio/region_mean": 0.0012068775482475757, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 716.0, |
| "completions/mean_length": 395.3399963378906, |
| "completions/mean_terminated_length": 361.61224365234375, |
| "completions/min_length": 135.0, |
| "completions/min_terminated_length": 135.0, |
| "entropy": 0.19389981627464295, |
| "epoch": 0.26492537313432835, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009714054875075817, |
| "learning_rate": 5e-05, |
| "loss": 0.0122, |
| "num_tokens": 1693373.0, |
| "reward": 0.2869628965854645, |
| "reward_std": 0.5468150973320007, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.19303710758686066, |
| "rewards/length_penalty/std": 0.1350608468055725, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927972555160522, |
| "sampling/importance_sampling_ratio/min": 0.13865187764167786, |
| "sampling/sampling_logp_difference/max": 1.975788950920105, |
| "sampling/sampling_logp_difference/mean": 0.01950252801179886, |
| "step": 71, |
| "step_time": 20.720340001862496 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010110618313774467, |
| "clip_ratio/high_mean": 0.0010110618313774467, |
| "clip_ratio/low_mean": 0.0002209374535595998, |
| "clip_ratio/low_min": 0.0002209374535595998, |
| "clip_ratio/region_mean": 0.0012319992762058972, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2010.0, |
| "completions/mean_length": 934.6799926757812, |
| "completions/mean_terminated_length": 690.2926635742188, |
| "completions/min_length": 148.0, |
| "completions/min_terminated_length": 148.0, |
| "entropy": 0.3338177978992462, |
| "epoch": 0.26865671641791045, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010647455230355263, |
| "learning_rate": 5e-05, |
| "loss": 0.0436, |
| "num_tokens": 1744897.0, |
| "reward": 0.06361328065395355, |
| "reward_std": 0.7316112518310547, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.4563867151737213, |
| "rewards/length_penalty/std": 0.36598309874534607, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9883248805999756, |
| "sampling/importance_sampling_ratio/min": 0.35565465688705444, |
| "sampling/sampling_logp_difference/max": 1.1421713829040527, |
| "sampling/sampling_logp_difference/mean": 0.021287694573402405, |
| "step": 72, |
| "step_time": 24.002578344894573 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015386869199573994, |
| "clip_ratio/high_mean": 0.00015386869199573994, |
| "clip_ratio/low_mean": 7.692307699471712e-05, |
| "clip_ratio/low_min": 7.692307699471712e-05, |
| "clip_ratio/region_mean": 0.00023079176899045705, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 493.0, |
| "completions/max_terminated_length": 493.0, |
| "completions/mean_length": 245.0800018310547, |
| "completions/mean_terminated_length": 245.0800018310547, |
| "completions/min_length": 62.0, |
| "completions/min_terminated_length": 62.0, |
| "entropy": 0.15735136568546296, |
| "epoch": 0.27238805970149255, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007244350388646126, |
| "learning_rate": 5e-05, |
| "loss": -0.0011, |
| "num_tokens": 1759941.0, |
| "reward": 0.84033203125, |
| "reward_std": 0.21418268978595734, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.11966796964406967, |
| "rewards/length_penalty/std": 0.04784605652093887, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9940878748893738, |
| "sampling/importance_sampling_ratio/min": 0.14858075976371765, |
| "sampling/sampling_logp_difference/max": 1.9066267013549805, |
| "sampling/sampling_logp_difference/mean": 0.020740604028105736, |
| "step": 73, |
| "step_time": 6.268040436087176 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047825013170950116, |
| "clip_ratio/high_mean": 0.00047825013170950116, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00047825013170950116, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 570.0, |
| "completions/max_terminated_length": 570.0, |
| "completions/mean_length": 218.17999267578125, |
| "completions/mean_terminated_length": 218.17999267578125, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.1820651412010193, |
| "epoch": 0.27611940298507465, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00898903887718916, |
| "learning_rate": 5e-05, |
| "loss": 0.0246, |
| "num_tokens": 1773100.0, |
| "reward": 0.8934667706489563, |
| "reward_std": 0.05137346312403679, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.10653319954872131, |
| "rewards/length_penalty/std": 0.05137345939874649, |
| "sampling/importance_sampling_ratio/max": 1.9743596315383911, |
| "sampling/importance_sampling_ratio/mean": 0.992275595664978, |
| "sampling/importance_sampling_ratio/min": 0.09230468422174454, |
| "sampling/sampling_logp_difference/max": 2.382660388946533, |
| "sampling/sampling_logp_difference/mean": 0.022644151002168655, |
| "step": 74, |
| "step_time": 6.304245180916041 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006831975246313959, |
| "clip_ratio/high_mean": 0.0006831975246313959, |
| "clip_ratio/low_mean": 0.00024684280942892655, |
| "clip_ratio/low_min": 0.00024684280942892655, |
| "clip_ratio/region_mean": 0.0009300403267843649, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1933.0, |
| "completions/mean_length": 648.4599609375, |
| "completions/mean_terminated_length": 526.7608642578125, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.31690038442611695, |
| "epoch": 0.2798507462686567, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014707234688103199, |
| "learning_rate": 5e-05, |
| "loss": 0.0169, |
| "num_tokens": 1809483.0, |
| "reward": 0.12336913496255875, |
| "reward_std": 0.7002291083335876, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.31663087010383606, |
| "rewards/length_penalty/std": 0.2994247078895569, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9892813563346863, |
| "sampling/importance_sampling_ratio/min": 0.2180526852607727, |
| "sampling/sampling_logp_difference/max": 1.5230185985565186, |
| "sampling/sampling_logp_difference/mean": 0.022157959640026093, |
| "step": 75, |
| "step_time": 22.700117832282558 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007681590621359646, |
| "clip_ratio/high_mean": 0.0007681590621359646, |
| "clip_ratio/low_mean": 0.0002631312469020486, |
| "clip_ratio/low_min": 0.0002631312469020486, |
| "clip_ratio/region_mean": 0.0010312903090380133, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 965.0, |
| "completions/max_terminated_length": 965.0, |
| "completions/mean_length": 287.8999938964844, |
| "completions/mean_terminated_length": 287.8999938964844, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.2121178150177002, |
| "epoch": 0.2835820895522388, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008129563182592392, |
| "learning_rate": 5e-05, |
| "loss": -0.0121, |
| "num_tokens": 1827468.0, |
| "reward": 0.4794238209724426, |
| "reward_std": 0.5341004133224487, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.14057616889476776, |
| "rewards/length_penalty/std": 0.07932731509208679, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9924062490463257, |
| "sampling/importance_sampling_ratio/min": 0.05626612529158592, |
| "sampling/sampling_logp_difference/max": 2.8776626586914062, |
| "sampling/sampling_logp_difference/mean": 0.021875016391277313, |
| "step": 76, |
| "step_time": 10.302457140060142 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006995088246185332, |
| "clip_ratio/high_mean": 0.0006995088246185332, |
| "clip_ratio/low_mean": 7.473310106433927e-05, |
| "clip_ratio/low_min": 7.473310106433927e-05, |
| "clip_ratio/region_mean": 0.0007742419315036386, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2041.0, |
| "completions/mean_length": 836.3999633789062, |
| "completions/mean_terminated_length": 701.7777709960938, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.3080302506685257, |
| "epoch": 0.2873134328358209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014947418123483658, |
| "learning_rate": 5e-05, |
| "loss": 0.0184, |
| "num_tokens": 1873568.0, |
| "reward": 0.2916015684604645, |
| "reward_std": 0.6831110715866089, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.40839844942092896, |
| "rewards/length_penalty/std": 0.3096544146537781, |
| "sampling/importance_sampling_ratio/max": 2.289604425430298, |
| "sampling/importance_sampling_ratio/mean": 0.9889733791351318, |
| "sampling/importance_sampling_ratio/min": 0.17478086054325104, |
| "sampling/sampling_logp_difference/max": 1.7442222833633423, |
| "sampling/sampling_logp_difference/mean": 0.020468981936573982, |
| "step": 77, |
| "step_time": 23.3045789769385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011105100275017321, |
| "clip_ratio/high_mean": 0.0011105100275017321, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011105100275017321, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 423.0, |
| "completions/max_terminated_length": 423.0, |
| "completions/mean_length": 196.95999145507812, |
| "completions/mean_terminated_length": 196.95999145507812, |
| "completions/min_length": 39.0, |
| "completions/min_terminated_length": 39.0, |
| "entropy": 0.17643568515777588, |
| "epoch": 0.291044776119403, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007783598266541958, |
| "learning_rate": 5e-05, |
| "loss": 0.0026, |
| "num_tokens": 1886246.0, |
| "reward": 0.6638281345367432, |
| "reward_std": 0.44404229521751404, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.09617187827825546, |
| "rewards/length_penalty/std": 0.046398621052503586, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929860830307007, |
| "sampling/importance_sampling_ratio/min": 0.02388361468911171, |
| "sampling/sampling_logp_difference/max": 3.734562635421753, |
| "sampling/sampling_logp_difference/mean": 0.02481072209775448, |
| "step": 78, |
| "step_time": 5.516294403932989 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014786585932597518, |
| "clip_ratio/high_mean": 0.0014786585932597518, |
| "clip_ratio/low_mean": 0.00026189438067376615, |
| "clip_ratio/low_min": 0.00026189438067376615, |
| "clip_ratio/region_mean": 0.0017405529273673893, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1569.0, |
| "completions/mean_length": 473.3399963378906, |
| "completions/mean_terminated_length": 407.72918701171875, |
| "completions/min_length": 39.0, |
| "completions/min_terminated_length": 39.0, |
| "entropy": 0.22272305488586425, |
| "epoch": 0.2947761194029851, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0138373002409935, |
| "learning_rate": 5e-05, |
| "loss": 0.0374, |
| "num_tokens": 1913823.0, |
| "reward": 0.6288769245147705, |
| "reward_std": 0.4718337059020996, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.23112304508686066, |
| "rewards/length_penalty/std": 0.25101178884506226, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9918286800384521, |
| "sampling/importance_sampling_ratio/min": 0.13701872527599335, |
| "sampling/sampling_logp_difference/max": 1.9876376390457153, |
| "sampling/sampling_logp_difference/mean": 0.019739534705877304, |
| "step": 79, |
| "step_time": 21.640338451368734 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018963391950819642, |
| "clip_ratio/high_mean": 0.0018963391950819642, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0018963391950819642, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 631.0, |
| "completions/max_terminated_length": 631.0, |
| "completions/mean_length": 199.6599884033203, |
| "completions/mean_terminated_length": 199.6599884033203, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.239141783118248, |
| "epoch": 0.29850746268656714, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008182297460734844, |
| "learning_rate": 5e-05, |
| "loss": 0.0016, |
| "num_tokens": 1927226.0, |
| "reward": 0.6825097799301147, |
| "reward_std": 0.4701237976551056, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.09749023616313934, |
| "rewards/length_penalty/std": 0.06250915676355362, |
| "sampling/importance_sampling_ratio/max": 2.9307050704956055, |
| "sampling/importance_sampling_ratio/mean": 0.991529643535614, |
| "sampling/importance_sampling_ratio/min": 0.2223271280527115, |
| "sampling/sampling_logp_difference/max": 1.5036054849624634, |
| "sampling/sampling_logp_difference/mean": 0.022932201623916626, |
| "step": 80, |
| "step_time": 7.0479226561728865 |
| }, |
| { |
| "clip_ratio/high_max": 0.000604342162841931, |
| "clip_ratio/high_mean": 0.000604342162841931, |
| "clip_ratio/low_mean": 6.788643368054182e-05, |
| "clip_ratio/low_min": 6.788643368054182e-05, |
| "clip_ratio/region_mean": 0.0006722285994328559, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1783.0, |
| "completions/mean_length": 444.67999267578125, |
| "completions/mean_terminated_length": 377.875, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.2760684221982956, |
| "epoch": 0.30223880597014924, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013124141842126846, |
| "learning_rate": 5e-05, |
| "loss": 0.0099, |
| "num_tokens": 1952930.0, |
| "reward": 0.3428710997104645, |
| "reward_std": 0.6616868376731873, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.2171289026737213, |
| "rewards/length_penalty/std": 0.25418272614479065, |
| "sampling/importance_sampling_ratio/max": 2.9184718132019043, |
| "sampling/importance_sampling_ratio/mean": 0.9895758628845215, |
| "sampling/importance_sampling_ratio/min": 0.04047023504972458, |
| "sampling/sampling_logp_difference/max": 3.207188606262207, |
| "sampling/sampling_logp_difference/mean": 0.024222921580076218, |
| "step": 81, |
| "step_time": 21.649420979898423 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006003081332892179, |
| "clip_ratio/high_mean": 0.0006003081332892179, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0006003081332892179, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 594.0, |
| "completions/max_terminated_length": 594.0, |
| "completions/mean_length": 184.13999938964844, |
| "completions/mean_terminated_length": 184.13999938964844, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.15603855848312378, |
| "epoch": 0.30597014925373134, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010622011497616768, |
| "learning_rate": 5e-05, |
| "loss": 0.0007, |
| "num_tokens": 1964717.0, |
| "reward": 0.6900878548622131, |
| "reward_std": 0.4795539975166321, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.08991210907697678, |
| "rewards/length_penalty/std": 0.07598428428173065, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9925838112831116, |
| "sampling/importance_sampling_ratio/min": 0.03661005198955536, |
| "sampling/sampling_logp_difference/max": 3.3074324131011963, |
| "sampling/sampling_logp_difference/mean": 0.025094155222177505, |
| "step": 82, |
| "step_time": 6.646573102101684 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009858429664745926, |
| "clip_ratio/high_mean": 0.0009858429664745926, |
| "clip_ratio/low_mean": 5.0075113540515305e-05, |
| "clip_ratio/low_min": 5.0075113540515305e-05, |
| "clip_ratio/region_mean": 0.001035918080015108, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 738.0, |
| "completions/max_terminated_length": 738.0, |
| "completions/mean_length": 331.94000244140625, |
| "completions/mean_terminated_length": 331.94000244140625, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.215926656126976, |
| "epoch": 0.30970149253731344, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009167241863906384, |
| "learning_rate": 5e-05, |
| "loss": -0.0168, |
| "num_tokens": 1983824.0, |
| "reward": 0.7979199290275574, |
| "reward_std": 0.20027850568294525, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.1620800793170929, |
| "rewards/length_penalty/std": 0.08078183978796005, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9915889501571655, |
| "sampling/importance_sampling_ratio/min": 0.205132856965065, |
| "sampling/sampling_logp_difference/max": 1.5840973854064941, |
| "sampling/sampling_logp_difference/mean": 0.01987045630812645, |
| "step": 83, |
| "step_time": 8.315541234100237 |
| }, |
| { |
| "clip_ratio/high_max": 0.002571542956866324, |
| "clip_ratio/high_mean": 0.002571542956866324, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.002571542956866324, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 228.0, |
| "completions/mean_length": 529.5199584960938, |
| "completions/mean_terminated_length": 149.90000915527344, |
| "completions/min_length": 48.0, |
| "completions/min_terminated_length": 48.0, |
| "entropy": 0.41015579700469973, |
| "epoch": 0.31343283582089554, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006999853998422623, |
| "learning_rate": 5e-05, |
| "loss": 0.0089, |
| "num_tokens": 2013090.0, |
| "reward": 0.34144529700279236, |
| "reward_std": 0.7737928628921509, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.25855469703674316, |
| "rewards/length_penalty/std": 0.3750412166118622, |
| "sampling/importance_sampling_ratio/max": 2.867025852203369, |
| "sampling/importance_sampling_ratio/mean": 0.9858042001724243, |
| "sampling/importance_sampling_ratio/min": 0.041259512305259705, |
| "sampling/sampling_logp_difference/max": 3.187873601913452, |
| "sampling/sampling_logp_difference/mean": 0.02589607611298561, |
| "step": 84, |
| "step_time": 22.430606939829886 |
| }, |
| { |
| "clip_ratio/high_max": 0.001008829683996737, |
| "clip_ratio/high_mean": 0.001008829683996737, |
| "clip_ratio/low_mean": 0.0002098641009069979, |
| "clip_ratio/low_min": 0.0002098641009069979, |
| "clip_ratio/region_mean": 0.001218693784903735, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1450.0, |
| "completions/max_terminated_length": 1450.0, |
| "completions/mean_length": 248.97999572753906, |
| "completions/mean_terminated_length": 248.97999572753906, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.21633342504501343, |
| "epoch": 0.31716417910447764, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00753357307985425, |
| "learning_rate": 5e-05, |
| "loss": 0.0005, |
| "num_tokens": 2028879.0, |
| "reward": 0.638427734375, |
| "reward_std": 0.4464810788631439, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.12157226353883743, |
| "rewards/length_penalty/std": 0.10566619038581848, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9920048117637634, |
| "sampling/importance_sampling_ratio/min": 0.21521638333797455, |
| "sampling/sampling_logp_difference/max": 1.8549904823303223, |
| "sampling/sampling_logp_difference/mean": 0.021208815276622772, |
| "step": 85, |
| "step_time": 14.633855456253514 |
| }, |
| { |
| "clip_ratio/high_max": 0.001333837426500395, |
| "clip_ratio/high_mean": 0.001333837426500395, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001333837426500395, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 495.0, |
| "completions/max_terminated_length": 495.0, |
| "completions/mean_length": 231.739990234375, |
| "completions/mean_terminated_length": 231.739990234375, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.17129210531711578, |
| "epoch": 0.3208955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008449839428067207, |
| "learning_rate": 5e-05, |
| "loss": 0.0079, |
| "num_tokens": 2042666.0, |
| "reward": 0.5268456935882568, |
| "reward_std": 0.4787832498550415, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.11315429955720901, |
| "rewards/length_penalty/std": 0.05298326164484024, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9937000274658203, |
| "sampling/importance_sampling_ratio/min": 0.08686614781618118, |
| "sampling/sampling_logp_difference/max": 2.4433867931365967, |
| "sampling/sampling_logp_difference/mean": 0.02323855087161064, |
| "step": 86, |
| "step_time": 5.850215919781476 |
| }, |
| { |
| "clip_ratio/high_max": 0.001132092683110386, |
| "clip_ratio/high_mean": 0.001132092683110386, |
| "clip_ratio/low_mean": 4.928536363877356e-05, |
| "clip_ratio/low_min": 4.928536363877356e-05, |
| "clip_ratio/region_mean": 0.0011813780525699257, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1513.0, |
| "completions/max_terminated_length": 1513.0, |
| "completions/mean_length": 339.17999267578125, |
| "completions/mean_terminated_length": 339.17999267578125, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.16397334337234498, |
| "epoch": 0.3246268656716418, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012496061623096466, |
| "learning_rate": 5e-05, |
| "loss": 0.0393, |
| "num_tokens": 2062525.0, |
| "reward": 0.794384777545929, |
| "reward_std": 0.27769869565963745, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.1656152307987213, |
| "rewards/length_penalty/std": 0.1305130273103714, |
| "sampling/importance_sampling_ratio/max": 2.4161219596862793, |
| "sampling/importance_sampling_ratio/mean": 0.993759036064148, |
| "sampling/importance_sampling_ratio/min": 0.02780325338244438, |
| "sampling/sampling_logp_difference/max": 3.5826022624969482, |
| "sampling/sampling_logp_difference/mean": 0.018865756690502167, |
| "step": 87, |
| "step_time": 15.59780843090266 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006534854124765843, |
| "clip_ratio/high_mean": 0.0006534854124765843, |
| "clip_ratio/low_mean": 0.00013055354065727443, |
| "clip_ratio/low_min": 0.00013055354065727443, |
| "clip_ratio/region_mean": 0.0007840389618650079, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1529.0, |
| "completions/mean_length": 489.2799987792969, |
| "completions/mean_terminated_length": 389.7872314453125, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.31865369975566865, |
| "epoch": 0.3283582089552239, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017925506457686424, |
| "learning_rate": 5e-05, |
| "loss": 0.0124, |
| "num_tokens": 2090599.0, |
| "reward": 0.4010937511920929, |
| "reward_std": 0.6635273098945618, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.23890624940395355, |
| "rewards/length_penalty/std": 0.26382654905319214, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9882604479789734, |
| "sampling/importance_sampling_ratio/min": 0.1960286945104599, |
| "sampling/sampling_logp_difference/max": 1.6294941902160645, |
| "sampling/sampling_logp_difference/mean": 0.023880664259195328, |
| "step": 88, |
| "step_time": 22.19856311706826 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007791827199980616, |
| "clip_ratio/high_mean": 0.0007791827199980616, |
| "clip_ratio/low_mean": 5.856515490449965e-05, |
| "clip_ratio/low_min": 5.856515490449965e-05, |
| "clip_ratio/region_mean": 0.0008377478690817953, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1266.0, |
| "completions/mean_length": 354.3800048828125, |
| "completions/mean_terminated_length": 319.8163146972656, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.22015844881534577, |
| "epoch": 0.332089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010026220232248306, |
| "learning_rate": 5e-05, |
| "loss": 0.0039, |
| "num_tokens": 2111378.0, |
| "reward": 0.6269629001617432, |
| "reward_std": 0.4989117681980133, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.1730371117591858, |
| "rewards/length_penalty/std": 0.16782726347446442, |
| "sampling/importance_sampling_ratio/max": 2.975550651550293, |
| "sampling/importance_sampling_ratio/mean": 0.9922510981559753, |
| "sampling/importance_sampling_ratio/min": 0.0389917828142643, |
| "sampling/sampling_logp_difference/max": 3.2444043159484863, |
| "sampling/sampling_logp_difference/mean": 0.020686915144324303, |
| "step": 89, |
| "step_time": 21.06516211712733 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011358758201822639, |
| "clip_ratio/high_mean": 0.0011358758201822639, |
| "clip_ratio/low_mean": 9.847366018220783e-05, |
| "clip_ratio/low_min": 9.847366018220783e-05, |
| "clip_ratio/region_mean": 0.001234349492006004, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 670.0, |
| "completions/max_terminated_length": 670.0, |
| "completions/mean_length": 215.72000122070312, |
| "completions/mean_terminated_length": 215.72000122070312, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.18611648082733154, |
| "epoch": 0.3358208955223881, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008773848414421082, |
| "learning_rate": 5e-05, |
| "loss": 0.0001, |
| "num_tokens": 2124444.0, |
| "reward": 0.6746679544448853, |
| "reward_std": 0.41931241750717163, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.10533203184604645, |
| "rewards/length_penalty/std": 0.04260782152414322, |
| "sampling/importance_sampling_ratio/max": 2.6770498752593994, |
| "sampling/importance_sampling_ratio/mean": 0.9935617446899414, |
| "sampling/importance_sampling_ratio/min": 0.04745025932788849, |
| "sampling/sampling_logp_difference/max": 3.0480732917785645, |
| "sampling/sampling_logp_difference/mean": 0.02464917115867138, |
| "step": 90, |
| "step_time": 7.1572527338285 |
| }, |
| { |
| "clip_ratio/high_max": 0.00110072148963809, |
| "clip_ratio/high_mean": 0.00110072148963809, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00110072148963809, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 296.0, |
| "completions/max_terminated_length": 296.0, |
| "completions/mean_length": 168.239990234375, |
| "completions/mean_terminated_length": 168.239990234375, |
| "completions/min_length": 82.0, |
| "completions/min_terminated_length": 82.0, |
| "entropy": 0.1710197865962982, |
| "epoch": 0.33955223880597013, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008983846753835678, |
| "learning_rate": 5e-05, |
| "loss": 0.007, |
| "num_tokens": 2135056.0, |
| "reward": 0.8978515267372131, |
| "reward_std": 0.15232199430465698, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.08214844018220901, |
| "rewards/length_penalty/std": 0.025607692077755928, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9940918684005737, |
| "sampling/importance_sampling_ratio/min": 0.10684461891651154, |
| "sampling/sampling_logp_difference/max": 2.236379623413086, |
| "sampling/sampling_logp_difference/mean": 0.02539275586605072, |
| "step": 91, |
| "step_time": 3.742940347176045 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010281951166689397, |
| "clip_ratio/high_mean": 0.0010281951166689397, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010281951166689397, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 604.0, |
| "completions/max_terminated_length": 604.0, |
| "completions/mean_length": 292.5, |
| "completions/mean_terminated_length": 292.5, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.1953798621892929, |
| "epoch": 0.34328358208955223, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007829510606825352, |
| "learning_rate": 5e-05, |
| "loss": 0.0003, |
| "num_tokens": 2153501.0, |
| "reward": 0.21717773377895355, |
| "reward_std": 0.4971000850200653, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.142822265625, |
| "rewards/length_penalty/std": 0.05672194063663483, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9931898713111877, |
| "sampling/importance_sampling_ratio/min": 0.13717234134674072, |
| "sampling/sampling_logp_difference/max": 1.9865171909332275, |
| "sampling/sampling_logp_difference/mean": 0.02159970812499523, |
| "step": 92, |
| "step_time": 7.300173272145912 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007858758908696473, |
| "clip_ratio/high_mean": 0.0007858758908696473, |
| "clip_ratio/low_mean": 0.00013071895809844137, |
| "clip_ratio/low_min": 0.00013071895809844137, |
| "clip_ratio/region_mean": 0.0009165948489680886, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 319.0, |
| "completions/max_terminated_length": 319.0, |
| "completions/mean_length": 172.01998901367188, |
| "completions/mean_terminated_length": 172.01998901367188, |
| "completions/min_length": 100.0, |
| "completions/min_terminated_length": 100.0, |
| "entropy": 0.18890211582183838, |
| "epoch": 0.34701492537313433, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007277416530996561, |
| "learning_rate": 5e-05, |
| "loss": 0.0039, |
| "num_tokens": 2165452.0, |
| "reward": 0.7560058236122131, |
| "reward_std": 0.3761473298072815, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.08399414271116257, |
| "rewards/length_penalty/std": 0.020517654716968536, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929280281066895, |
| "sampling/importance_sampling_ratio/min": 0.05409553647041321, |
| "sampling/sampling_logp_difference/max": 2.917003631591797, |
| "sampling/sampling_logp_difference/mean": 0.0271185040473938, |
| "step": 93, |
| "step_time": 4.145558499963954 |
| }, |
| { |
| "clip_ratio/high_max": 0.000785779458237812, |
| "clip_ratio/high_mean": 0.000785779458237812, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000785779458237812, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1327.0, |
| "completions/max_terminated_length": 1327.0, |
| "completions/mean_length": 377.5799865722656, |
| "completions/mean_terminated_length": 377.5799865722656, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.22692514657974244, |
| "epoch": 0.35074626865671643, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013048828579485416, |
| "learning_rate": 5e-05, |
| "loss": -0.0065, |
| "num_tokens": 2187441.0, |
| "reward": 0.09563476592302322, |
| "reward_std": 0.48391589522361755, |
| "rewards/correctness/mean": 0.2800000011920929, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.18436522781848907, |
| "rewards/length_penalty/std": 0.11944812536239624, |
| "sampling/importance_sampling_ratio/max": 2.2977609634399414, |
| "sampling/importance_sampling_ratio/mean": 0.9918870329856873, |
| "sampling/importance_sampling_ratio/min": 0.17901191115379333, |
| "sampling/sampling_logp_difference/max": 1.720302939414978, |
| "sampling/sampling_logp_difference/mean": 0.01977582275867462, |
| "step": 94, |
| "step_time": 14.208001327002421 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003195017226971686, |
| "clip_ratio/high_mean": 0.0003195017226971686, |
| "clip_ratio/low_mean": 0.0001664841634919867, |
| "clip_ratio/low_min": 0.0001664841634919867, |
| "clip_ratio/region_mean": 0.0004859858890995383, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 895.0, |
| "completions/mean_length": 515.1799926757812, |
| "completions/mean_terminated_length": 223.21429443359375, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.14921582788228988, |
| "epoch": 0.35447761194029853, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009451565332710743, |
| "learning_rate": 5e-05, |
| "loss": 0.0568, |
| "num_tokens": 2216610.0, |
| "reward": 0.5484472513198853, |
| "reward_std": 0.7319607734680176, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.2515527307987213, |
| "rewards/length_penalty/std": 0.33770838379859924, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9945523738861084, |
| "sampling/importance_sampling_ratio/min": 0.15370461344718933, |
| "sampling/sampling_logp_difference/max": 1.8727226257324219, |
| "sampling/sampling_logp_difference/mean": 0.012529810890555382, |
| "step": 95, |
| "step_time": 22.457538310205564 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015879049897193909, |
| "clip_ratio/high_mean": 0.0015879049897193909, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0015879049897193909, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 693.0, |
| "completions/max_terminated_length": 693.0, |
| "completions/mean_length": 257.3800048828125, |
| "completions/mean_terminated_length": 257.3800048828125, |
| "completions/min_length": 53.0, |
| "completions/min_terminated_length": 53.0, |
| "entropy": 0.19005478024482728, |
| "epoch": 0.3582089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014516693539917469, |
| "learning_rate": 5e-05, |
| "loss": -0.0002, |
| "num_tokens": 2234099.0, |
| "reward": 0.634326159954071, |
| "reward_std": 0.4057468771934509, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.1256738305091858, |
| "rewards/length_penalty/std": 0.08897071331739426, |
| "sampling/importance_sampling_ratio/max": 2.5535054206848145, |
| "sampling/importance_sampling_ratio/mean": 0.9931633472442627, |
| "sampling/importance_sampling_ratio/min": 0.1190047562122345, |
| "sampling/sampling_logp_difference/max": 2.128591775894165, |
| "sampling/sampling_logp_difference/mean": 0.019039038568735123, |
| "step": 96, |
| "step_time": 8.3912249119021 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006516657507745549, |
| "clip_ratio/high_mean": 0.0006516657507745549, |
| "clip_ratio/low_mean": 7.738037384115159e-05, |
| "clip_ratio/low_min": 7.738037384115159e-05, |
| "clip_ratio/region_mean": 0.0007290461275260895, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2023.0, |
| "completions/mean_length": 530.4599609375, |
| "completions/mean_terminated_length": 361.8444519042969, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.18799397051334382, |
| "epoch": 0.3619402985074627, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013441347517073154, |
| "learning_rate": 5e-05, |
| "loss": 0.0231, |
| "num_tokens": 2263752.0, |
| "reward": 0.3809863328933716, |
| "reward_std": 0.6652371287345886, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.25901368260383606, |
| "rewards/length_penalty/std": 0.32686105370521545, |
| "sampling/importance_sampling_ratio/max": 2.4711928367614746, |
| "sampling/importance_sampling_ratio/mean": 0.9931226968765259, |
| "sampling/importance_sampling_ratio/min": 0.15734703838825226, |
| "sampling/sampling_logp_difference/max": 1.8493014574050903, |
| "sampling/sampling_logp_difference/mean": 0.01476562675088644, |
| "step": 97, |
| "step_time": 22.094223432941362 |
| }, |
| { |
| "clip_ratio/high_max": 0.00079935536487028, |
| "clip_ratio/high_mean": 0.00079935536487028, |
| "clip_ratio/low_mean": 0.00012242774537298828, |
| "clip_ratio/low_min": 0.00012242774537298828, |
| "clip_ratio/region_mean": 0.0009217831102432683, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1552.0, |
| "completions/max_terminated_length": 1552.0, |
| "completions/mean_length": 400.03997802734375, |
| "completions/mean_terminated_length": 400.03997802734375, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.2498596727848053, |
| "epoch": 0.3656716417910448, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012937057763338089, |
| "learning_rate": 5e-05, |
| "loss": -0.0175, |
| "num_tokens": 2287024.0, |
| "reward": 0.6446679830551147, |
| "reward_std": 0.5091517567634583, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.19533203542232513, |
| "rewards/length_penalty/std": 0.19338354468345642, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9906523823738098, |
| "sampling/importance_sampling_ratio/min": 0.288949191570282, |
| "sampling/sampling_logp_difference/max": 1.241504430770874, |
| "sampling/sampling_logp_difference/mean": 0.02085532620549202, |
| "step": 98, |
| "step_time": 16.582799996016547 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012652603443711996, |
| "clip_ratio/high_mean": 0.0012652603443711996, |
| "clip_ratio/low_mean": 0.00011074523790739476, |
| "clip_ratio/low_min": 0.00011074523790739476, |
| "clip_ratio/region_mean": 0.0013760055531747638, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 741.0, |
| "completions/max_terminated_length": 741.0, |
| "completions/mean_length": 333.67999267578125, |
| "completions/mean_terminated_length": 333.67999267578125, |
| "completions/min_length": 134.0, |
| "completions/min_terminated_length": 134.0, |
| "entropy": 0.24855854511260986, |
| "epoch": 0.3694029850746269, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0096174580976367, |
| "learning_rate": 5e-05, |
| "loss": 0.007, |
| "num_tokens": 2306978.0, |
| "reward": 0.3370703160762787, |
| "reward_std": 0.5478963255882263, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.1629296839237213, |
| "rewards/length_penalty/std": 0.0706363394856453, |
| "sampling/importance_sampling_ratio/max": 2.512291669845581, |
| "sampling/importance_sampling_ratio/mean": 0.9905197024345398, |
| "sampling/importance_sampling_ratio/min": 0.21949462592601776, |
| "sampling/sampling_logp_difference/max": 1.5164275169372559, |
| "sampling/sampling_logp_difference/mean": 0.021251078695058823, |
| "step": 99, |
| "step_time": 8.588250870350748 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011286438559181989, |
| "clip_ratio/high_mean": 0.0011286438559181989, |
| "clip_ratio/low_mean": 0.0003009781707078218, |
| "clip_ratio/low_min": 0.0003009781707078218, |
| "clip_ratio/region_mean": 0.001429622049909085, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 389.0, |
| "completions/max_terminated_length": 389.0, |
| "completions/mean_length": 161.3000030517578, |
| "completions/mean_terminated_length": 161.3000030517578, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.16090967059135436, |
| "epoch": 0.373134328358209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006327748764306307, |
| "learning_rate": 5e-05, |
| "loss": 0.0086, |
| "num_tokens": 2317703.0, |
| "reward": 0.7012402415275574, |
| "reward_std": 0.43009650707244873, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.07875976711511612, |
| "rewards/length_penalty/std": 0.03283224254846573, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9934931993484497, |
| "sampling/importance_sampling_ratio/min": 0.17191310226917267, |
| "sampling/sampling_logp_difference/max": 1.7607661485671997, |
| "sampling/sampling_logp_difference/mean": 0.020073654130101204, |
| "step": 100, |
| "step_time": 4.7784998482093215 |
| }, |
| { |
| "clip_ratio/high_max": 0.00014534431393258273, |
| "clip_ratio/high_mean": 0.00014534431393258273, |
| "clip_ratio/low_mean": 4.60087409010157e-05, |
| "clip_ratio/low_min": 4.60087409010157e-05, |
| "clip_ratio/region_mean": 0.00019135305483359845, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1737.0, |
| "completions/mean_length": 316.3399963378906, |
| "completions/mean_terminated_length": 281.0, |
| "completions/min_length": 66.0, |
| "completions/min_terminated_length": 66.0, |
| "entropy": 0.2870136916637421, |
| "epoch": 0.376865671641791, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01009181048721075, |
| "learning_rate": 5e-05, |
| "loss": 0.0499, |
| "num_tokens": 2336090.0, |
| "reward": 0.46553710103034973, |
| "reward_std": 0.5960493683815002, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.15446288883686066, |
| "rewards/length_penalty/std": 0.19165311753749847, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9892403483390808, |
| "sampling/importance_sampling_ratio/min": 0.34832897782325745, |
| "sampling/sampling_logp_difference/max": 1.1428550481796265, |
| "sampling/sampling_logp_difference/mean": 0.02269444428384304, |
| "step": 101, |
| "step_time": 21.316057774005458 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013752088241744786, |
| "clip_ratio/high_mean": 0.0013752088241744786, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0013752088241744786, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 430.0, |
| "completions/max_terminated_length": 430.0, |
| "completions/mean_length": 211.3199920654297, |
| "completions/mean_terminated_length": 211.3199920654297, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.1985709100961685, |
| "epoch": 0.3805970149253731, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00852118618786335, |
| "learning_rate": 5e-05, |
| "loss": -0.0004, |
| "num_tokens": 2349396.0, |
| "reward": 0.8568164110183716, |
| "reward_std": 0.21128198504447937, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.10318359732627869, |
| "rewards/length_penalty/std": 0.04691115394234657, |
| "sampling/importance_sampling_ratio/max": 2.41068959236145, |
| "sampling/importance_sampling_ratio/mean": 0.9922513961791992, |
| "sampling/importance_sampling_ratio/min": 0.19981122016906738, |
| "sampling/sampling_logp_difference/max": 1.610382318496704, |
| "sampling/sampling_logp_difference/mean": 0.021990343928337097, |
| "step": 102, |
| "step_time": 5.150301951915026 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010086591704748572, |
| "clip_ratio/high_mean": 0.0010086591704748572, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010086591704748572, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 552.0, |
| "completions/max_terminated_length": 552.0, |
| "completions/mean_length": 261.3599853515625, |
| "completions/mean_terminated_length": 261.3599853515625, |
| "completions/min_length": 136.0, |
| "completions/min_terminated_length": 136.0, |
| "entropy": 0.20639686584472655, |
| "epoch": 0.3843283582089552, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00844596792012453, |
| "learning_rate": 5e-05, |
| "loss": 0.0198, |
| "num_tokens": 2365234.0, |
| "reward": 0.8723828196525574, |
| "reward_std": 0.050984714180231094, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.12761718034744263, |
| "rewards/length_penalty/std": 0.050984714180231094, |
| "sampling/importance_sampling_ratio/max": 2.239100694656372, |
| "sampling/importance_sampling_ratio/mean": 0.9925338625907898, |
| "sampling/importance_sampling_ratio/min": 0.2017344981431961, |
| "sampling/sampling_logp_difference/max": 1.6008027791976929, |
| "sampling/sampling_logp_difference/mean": 0.021596092730760574, |
| "step": 103, |
| "step_time": 6.470795038854703 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012988864706130697, |
| "clip_ratio/high_mean": 0.0012988864706130697, |
| "clip_ratio/low_mean": 0.00029099011735524984, |
| "clip_ratio/low_min": 0.00029099011735524984, |
| "clip_ratio/region_mean": 0.0015898765821475535, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1836.0, |
| "completions/mean_length": 511.05999755859375, |
| "completions/mean_terminated_length": 479.6938781738281, |
| "completions/min_length": 141.0, |
| "completions/min_terminated_length": 141.0, |
| "entropy": 0.2932967752218246, |
| "epoch": 0.3880597014925373, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010907494463026524, |
| "learning_rate": 5e-05, |
| "loss": -0.0347, |
| "num_tokens": 2394687.0, |
| "reward": 0.39045897126197815, |
| "reward_std": 0.6349729895591736, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.2495410144329071, |
| "rewards/length_penalty/std": 0.22605769336223602, |
| "sampling/importance_sampling_ratio/max": 1.9733953475952148, |
| "sampling/importance_sampling_ratio/mean": 0.988997220993042, |
| "sampling/importance_sampling_ratio/min": 0.12278058379888535, |
| "sampling/sampling_logp_difference/max": 2.0973563194274902, |
| "sampling/sampling_logp_difference/mean": 0.02159745618700981, |
| "step": 104, |
| "step_time": 22.426975200651214 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012662688503041863, |
| "clip_ratio/high_mean": 0.0012662688503041863, |
| "clip_ratio/low_mean": 0.0001231383066624403, |
| "clip_ratio/low_min": 0.0001231383066624403, |
| "clip_ratio/region_mean": 0.0013894071686081587, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1788.0, |
| "completions/mean_length": 487.91998291015625, |
| "completions/mean_terminated_length": 388.3404235839844, |
| "completions/min_length": 96.0, |
| "completions/min_terminated_length": 96.0, |
| "entropy": 0.33756410479545595, |
| "epoch": 0.3917910447761194, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011040698736906052, |
| "learning_rate": 5e-05, |
| "loss": 0.0098, |
| "num_tokens": 2421723.0, |
| "reward": 0.20175780355930328, |
| "reward_std": 0.6327792406082153, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.23824219405651093, |
| "rewards/length_penalty/std": 0.2588142156600952, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9877537488937378, |
| "sampling/importance_sampling_ratio/min": 0.22850362956523895, |
| "sampling/sampling_logp_difference/max": 1.5152201652526855, |
| "sampling/sampling_logp_difference/mean": 0.02386472560465336, |
| "step": 105, |
| "step_time": 21.693851647898555 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007058828137814999, |
| "clip_ratio/high_mean": 0.0007058828137814999, |
| "clip_ratio/low_mean": 0.0002176596492063254, |
| "clip_ratio/low_min": 0.0002176596492063254, |
| "clip_ratio/region_mean": 0.0009235424571670592, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 532.0, |
| "completions/max_terminated_length": 532.0, |
| "completions/mean_length": 274.44000244140625, |
| "completions/mean_terminated_length": 274.44000244140625, |
| "completions/min_length": 137.0, |
| "completions/min_terminated_length": 137.0, |
| "entropy": 0.18638840913772584, |
| "epoch": 0.39552238805970147, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007653559558093548, |
| "learning_rate": 5e-05, |
| "loss": -0.0044, |
| "num_tokens": 2438955.0, |
| "reward": 0.7059960961341858, |
| "reward_std": 0.36752375960350037, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.1340039074420929, |
| "rewards/length_penalty/std": 0.05161964148283005, |
| "sampling/importance_sampling_ratio/max": 2.965254068374634, |
| "sampling/importance_sampling_ratio/mean": 0.9933432340621948, |
| "sampling/importance_sampling_ratio/min": 0.13162203133106232, |
| "sampling/sampling_logp_difference/max": 2.0278208255767822, |
| "sampling/sampling_logp_difference/mean": 0.019450847059488297, |
| "step": 106, |
| "step_time": 6.528385950950906 |
| }, |
| { |
| "clip_ratio/high_max": 0.00037091612175572665, |
| "clip_ratio/high_mean": 0.00037091612175572665, |
| "clip_ratio/low_mean": 0.00022797733545303344, |
| "clip_ratio/low_min": 0.00022797733545303344, |
| "clip_ratio/region_mean": 0.0005988934659399092, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1530.0, |
| "completions/mean_length": 387.239990234375, |
| "completions/mean_terminated_length": 353.346923828125, |
| "completions/min_length": 101.0, |
| "completions/min_terminated_length": 101.0, |
| "entropy": 0.20107284784317017, |
| "epoch": 0.39925373134328357, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011592432856559753, |
| "learning_rate": 5e-05, |
| "loss": 0.0231, |
| "num_tokens": 2461307.0, |
| "reward": 0.6309179663658142, |
| "reward_std": 0.5046394467353821, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.18908202648162842, |
| "rewards/length_penalty/std": 0.17699696123600006, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929824471473694, |
| "sampling/importance_sampling_ratio/min": 0.2182321846485138, |
| "sampling/sampling_logp_difference/max": 1.5221956968307495, |
| "sampling/sampling_logp_difference/mean": 0.017658578231930733, |
| "step": 107, |
| "step_time": 21.40512843313627 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010892539867199957, |
| "clip_ratio/high_mean": 0.0010892539867199957, |
| "clip_ratio/low_mean": 0.00025998064666055143, |
| "clip_ratio/low_min": 0.00025998064666055143, |
| "clip_ratio/region_mean": 0.001349234627559781, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 860.0, |
| "completions/max_terminated_length": 860.0, |
| "completions/mean_length": 230.36000061035156, |
| "completions/mean_terminated_length": 230.36000061035156, |
| "completions/min_length": 77.0, |
| "completions/min_terminated_length": 77.0, |
| "entropy": 0.24926120042800903, |
| "epoch": 0.40298507462686567, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009040621109306812, |
| "learning_rate": 5e-05, |
| "loss": -0.0111, |
| "num_tokens": 2475465.0, |
| "reward": 0.7475194931030273, |
| "reward_std": 0.3557127118110657, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.11248046904802322, |
| "rewards/length_penalty/std": 0.07028742879629135, |
| "sampling/importance_sampling_ratio/max": 2.269204616546631, |
| "sampling/importance_sampling_ratio/mean": 0.9903947710990906, |
| "sampling/importance_sampling_ratio/min": 0.1813625693321228, |
| "sampling/sampling_logp_difference/max": 1.7072571516036987, |
| "sampling/sampling_logp_difference/mean": 0.024531563743948936, |
| "step": 108, |
| "step_time": 9.08179205446504 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009695745538920164, |
| "clip_ratio/high_mean": 0.0009695745538920164, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0009695745538920164, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 346.0, |
| "completions/max_terminated_length": 346.0, |
| "completions/mean_length": 226.79998779296875, |
| "completions/mean_terminated_length": 226.79998779296875, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.18980790674686432, |
| "epoch": 0.40671641791044777, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008144025690853596, |
| "learning_rate": 5e-05, |
| "loss": 0.0006, |
| "num_tokens": 2490025.0, |
| "reward": 0.4492577910423279, |
| "reward_std": 0.49622607231140137, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.11074218899011612, |
| "rewards/length_penalty/std": 0.022443166002631187, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9919015169143677, |
| "sampling/importance_sampling_ratio/min": 0.1187991201877594, |
| "sampling/sampling_logp_difference/max": 2.1303212642669678, |
| "sampling/sampling_logp_difference/mean": 0.02290290966629982, |
| "step": 109, |
| "step_time": 4.580782901961356 |
| }, |
| { |
| "clip_ratio/high_max": 0.001082241104450077, |
| "clip_ratio/high_mean": 0.001082241104450077, |
| "clip_ratio/low_mean": 0.00011204482289031147, |
| "clip_ratio/low_min": 0.00011204482289031147, |
| "clip_ratio/region_mean": 0.0011942859389819204, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 407.0, |
| "completions/max_terminated_length": 407.0, |
| "completions/mean_length": 178.01998901367188, |
| "completions/mean_terminated_length": 178.01998901367188, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.20231384932994842, |
| "epoch": 0.41044776119402987, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008012822829186916, |
| "learning_rate": 5e-05, |
| "loss": 0.0138, |
| "num_tokens": 2502006.0, |
| "reward": 0.9130761623382568, |
| "reward_std": 0.040956564247608185, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.08692383021116257, |
| "rewards/length_penalty/std": 0.040956564247608185, |
| "sampling/importance_sampling_ratio/max": 2.4998886585235596, |
| "sampling/importance_sampling_ratio/mean": 0.9920247793197632, |
| "sampling/importance_sampling_ratio/min": 0.0835471898317337, |
| "sampling/sampling_logp_difference/max": 2.4823436737060547, |
| "sampling/sampling_logp_difference/mean": 0.027247637510299683, |
| "step": 110, |
| "step_time": 4.893668755656108 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014011939289048313, |
| "clip_ratio/high_mean": 0.0014011939289048313, |
| "clip_ratio/low_mean": 0.0002211142418673262, |
| "clip_ratio/low_min": 0.0002211142418673262, |
| "clip_ratio/region_mean": 0.0016223081853240728, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1160.0, |
| "completions/mean_length": 379.9599914550781, |
| "completions/mean_terminated_length": 345.9183654785156, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.2529039353132248, |
| "epoch": 0.4141791044776119, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012225795537233353, |
| "learning_rate": 5e-05, |
| "loss": 0.0179, |
| "num_tokens": 2525384.0, |
| "reward": 0.634472668170929, |
| "reward_std": 0.48677435517311096, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.18552733957767487, |
| "rewards/length_penalty/std": 0.18833178281784058, |
| "sampling/importance_sampling_ratio/max": 2.1832826137542725, |
| "sampling/importance_sampling_ratio/mean": 0.9897897243499756, |
| "sampling/importance_sampling_ratio/min": 0.05665816366672516, |
| "sampling/sampling_logp_difference/max": 2.8707191944122314, |
| "sampling/sampling_logp_difference/mean": 0.021677521988749504, |
| "step": 111, |
| "step_time": 21.17579821124673 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006945164932403713, |
| "clip_ratio/high_mean": 0.0006945164932403713, |
| "clip_ratio/low_mean": 5.104645388200879e-05, |
| "clip_ratio/low_min": 5.104645388200879e-05, |
| "clip_ratio/region_mean": 0.0007455629471223802, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1024.0, |
| "completions/max_terminated_length": 1024.0, |
| "completions/mean_length": 308.5199890136719, |
| "completions/mean_terminated_length": 308.5199890136719, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.2467365562915802, |
| "epoch": 0.417910447761194, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011095298454165459, |
| "learning_rate": 5e-05, |
| "loss": -0.0059, |
| "num_tokens": 2543770.0, |
| "reward": 0.44935545325279236, |
| "reward_std": 0.5310330390930176, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.15064452588558197, |
| "rewards/length_penalty/std": 0.11441416293382645, |
| "sampling/importance_sampling_ratio/max": 2.491806745529175, |
| "sampling/importance_sampling_ratio/mean": 0.9903685450553894, |
| "sampling/importance_sampling_ratio/min": 0.03680182248353958, |
| "sampling/sampling_logp_difference/max": 3.3022079467773438, |
| "sampling/sampling_logp_difference/mean": 0.023889312520623207, |
| "step": 112, |
| "step_time": 11.023793266154826 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008405714528635144, |
| "clip_ratio/high_mean": 0.0008405714528635144, |
| "clip_ratio/low_mean": 0.0001233021728694439, |
| "clip_ratio/low_min": 0.0001233021728694439, |
| "clip_ratio/region_mean": 0.0009638736257329584, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 699.0, |
| "completions/max_terminated_length": 699.0, |
| "completions/mean_length": 296.6000061035156, |
| "completions/mean_terminated_length": 296.6000061035156, |
| "completions/min_length": 162.0, |
| "completions/min_terminated_length": 162.0, |
| "entropy": 0.2061973661184311, |
| "epoch": 0.4216417910447761, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011749361641705036, |
| "learning_rate": 5e-05, |
| "loss": 0.0067, |
| "num_tokens": 2561860.0, |
| "reward": 0.5951757431030273, |
| "reward_std": 0.4404974579811096, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.14482422173023224, |
| "rewards/length_penalty/std": 0.061369024217128754, |
| "sampling/importance_sampling_ratio/max": 2.9565534591674805, |
| "sampling/importance_sampling_ratio/mean": 0.9917761087417603, |
| "sampling/importance_sampling_ratio/min": 0.030206363648176193, |
| "sampling/sampling_logp_difference/max": 3.4997026920318604, |
| "sampling/sampling_logp_difference/mean": 0.02093423157930374, |
| "step": 113, |
| "step_time": 8.319697152124718 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003970591991674155, |
| "clip_ratio/high_mean": 0.0003970591991674155, |
| "clip_ratio/low_mean": 0.00018602642230689526, |
| "clip_ratio/low_min": 0.00018602642230689526, |
| "clip_ratio/region_mean": 0.0005830856185639277, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1333.0, |
| "completions/mean_length": 643.9400024414062, |
| "completions/mean_terminated_length": 335.731689453125, |
| "completions/min_length": 155.0, |
| "completions/min_terminated_length": 155.0, |
| "entropy": 0.3869974434375763, |
| "epoch": 0.4253731343283582, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010365336202085018, |
| "learning_rate": 5e-05, |
| "loss": -0.013, |
| "num_tokens": 2598907.0, |
| "reward": 0.04557617008686066, |
| "reward_std": 0.6788390278816223, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.3144238293170929, |
| "rewards/length_penalty/std": 0.34203028678894043, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9866120219230652, |
| "sampling/importance_sampling_ratio/min": 0.2539859712123871, |
| "sampling/sampling_logp_difference/max": 1.370476245880127, |
| "sampling/sampling_logp_difference/mean": 0.025814063847064972, |
| "step": 114, |
| "step_time": 22.932738778181374 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014580443385057151, |
| "clip_ratio/high_mean": 0.0014580443385057151, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0014580443385057151, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 817.0, |
| "completions/max_terminated_length": 817.0, |
| "completions/mean_length": 178.0800018310547, |
| "completions/mean_terminated_length": 178.0800018310547, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.19586332738399506, |
| "epoch": 0.4291044776119403, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011998329311609268, |
| "learning_rate": 5e-05, |
| "loss": 0.0389, |
| "num_tokens": 2610721.0, |
| "reward": 0.8930468559265137, |
| "reward_std": 0.18783065676689148, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.08695312589406967, |
| "rewards/length_penalty/std": 0.05046559125185013, |
| "sampling/importance_sampling_ratio/max": 1.859188437461853, |
| "sampling/importance_sampling_ratio/mean": 0.9927067160606384, |
| "sampling/importance_sampling_ratio/min": 0.031091347336769104, |
| "sampling/sampling_logp_difference/max": 3.470825672149658, |
| "sampling/sampling_logp_difference/mean": 0.021583182737231255, |
| "step": 115, |
| "step_time": 8.240436677122489 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010104132466949522, |
| "clip_ratio/high_mean": 0.0010104132466949522, |
| "clip_ratio/low_mean": 0.00013272052747197448, |
| "clip_ratio/low_min": 0.00013272052747197448, |
| "clip_ratio/region_mean": 0.001143133791629225, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1737.0, |
| "completions/max_terminated_length": 1737.0, |
| "completions/mean_length": 334.0199890136719, |
| "completions/mean_terminated_length": 334.0199890136719, |
| "completions/min_length": 77.0, |
| "completions/min_terminated_length": 77.0, |
| "entropy": 0.4037789165973663, |
| "epoch": 0.43283582089552236, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011798610910773277, |
| "learning_rate": 5e-05, |
| "loss": 0.0432, |
| "num_tokens": 2630872.0, |
| "reward": 0.43690428137779236, |
| "reward_std": 0.6396973729133606, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.16309569776058197, |
| "rewards/length_penalty/std": 0.21199752390384674, |
| "sampling/importance_sampling_ratio/max": 2.6321260929107666, |
| "sampling/importance_sampling_ratio/mean": 0.9858199954032898, |
| "sampling/importance_sampling_ratio/min": 0.2427348643541336, |
| "sampling/sampling_logp_difference/max": 1.415785551071167, |
| "sampling/sampling_logp_difference/mean": 0.028386462479829788, |
| "step": 116, |
| "step_time": 18.163943129358813 |
| }, |
| { |
| "clip_ratio/high_max": 0.001447824144270271, |
| "clip_ratio/high_mean": 0.001447824144270271, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001447824144270271, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 857.0, |
| "completions/max_terminated_length": 857.0, |
| "completions/mean_length": 253.739990234375, |
| "completions/mean_terminated_length": 253.739990234375, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.23947942554950713, |
| "epoch": 0.43656716417910446, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011144926771521568, |
| "learning_rate": 5e-05, |
| "loss": -0.0043, |
| "num_tokens": 2645939.0, |
| "reward": 0.5961034893989563, |
| "reward_std": 0.45136216282844543, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.12389648705720901, |
| "rewards/length_penalty/std": 0.07663574069738388, |
| "sampling/importance_sampling_ratio/max": 2.471953868865967, |
| "sampling/importance_sampling_ratio/mean": 0.9911974668502808, |
| "sampling/importance_sampling_ratio/min": 0.013607542961835861, |
| "sampling/sampling_logp_difference/max": 4.297131061553955, |
| "sampling/sampling_logp_difference/mean": 0.022656066343188286, |
| "step": 117, |
| "step_time": 9.087454076623544 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009678983653429896, |
| "clip_ratio/high_mean": 0.0009678983653429896, |
| "clip_ratio/low_mean": 0.000283385266084224, |
| "clip_ratio/low_min": 0.000283385266084224, |
| "clip_ratio/region_mean": 0.0012512836256064475, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1605.0, |
| "completions/mean_length": 516.3599853515625, |
| "completions/mean_terminated_length": 307.5, |
| "completions/min_length": 87.0, |
| "completions/min_terminated_length": 87.0, |
| "entropy": 0.3771422505378723, |
| "epoch": 0.44029850746268656, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008571499027311802, |
| "learning_rate": 5e-05, |
| "loss": 0.0013, |
| "num_tokens": 2675357.0, |
| "reward": 0.24787108600139618, |
| "reward_std": 0.7042934894561768, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.2521288990974426, |
| "rewards/length_penalty/std": 0.30630889534950256, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9865365624427795, |
| "sampling/importance_sampling_ratio/min": 0.01634548045694828, |
| "sampling/sampling_logp_difference/max": 4.113803863525391, |
| "sampling/sampling_logp_difference/mean": 0.025905821472406387, |
| "step": 118, |
| "step_time": 22.536475989967585 |
| }, |
| { |
| "clip_ratio/high_max": 0.002697722613811493, |
| "clip_ratio/high_mean": 0.002697722613811493, |
| "clip_ratio/low_mean": 0.00014847810380160808, |
| "clip_ratio/low_min": 0.00014847810380160808, |
| "clip_ratio/region_mean": 0.002846200717613101, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 547.0, |
| "completions/max_terminated_length": 547.0, |
| "completions/mean_length": 151.0399932861328, |
| "completions/mean_terminated_length": 151.0399932861328, |
| "completions/min_length": 68.0, |
| "completions/min_terminated_length": 68.0, |
| "entropy": 0.23717451989650726, |
| "epoch": 0.44402985074626866, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007095003966242075, |
| "learning_rate": 5e-05, |
| "loss": 0.0002, |
| "num_tokens": 2684739.0, |
| "reward": 0.8662499785423279, |
| "reward_std": 0.24807299673557281, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.07374999672174454, |
| "rewards/length_penalty/std": 0.03579280525445938, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9917730689048767, |
| "sampling/importance_sampling_ratio/min": 0.10631968080997467, |
| "sampling/sampling_logp_difference/max": 2.241304874420166, |
| "sampling/sampling_logp_difference/mean": 0.024992216378450394, |
| "step": 119, |
| "step_time": 5.7867826006840914 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007599756674608216, |
| "clip_ratio/high_mean": 0.0007599756674608216, |
| "clip_ratio/low_mean": 0.00025227312289644033, |
| "clip_ratio/low_min": 0.00025227312289644033, |
| "clip_ratio/region_mean": 0.001012248790357262, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1382.0, |
| "completions/mean_length": 563.6599731445312, |
| "completions/mean_terminated_length": 237.82925415039062, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.25614960491657257, |
| "epoch": 0.44776119402985076, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0075630908831954, |
| "learning_rate": 5e-05, |
| "loss": 0.0013, |
| "num_tokens": 2716202.0, |
| "reward": 0.10477539151906967, |
| "reward_std": 0.7107500433921814, |
| "rewards/correctness/mean": 0.3799999952316284, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.27522459626197815, |
| "rewards/length_penalty/std": 0.35428354144096375, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9896450042724609, |
| "sampling/importance_sampling_ratio/min": 0.036172062158584595, |
| "sampling/sampling_logp_difference/max": 3.3194682598114014, |
| "sampling/sampling_logp_difference/mean": 0.019285691902041435, |
| "step": 120, |
| "step_time": 22.28823065175675 |
| }, |
| { |
| "clip_ratio/high_max": 0.001447782351169735, |
| "clip_ratio/high_mean": 0.001447782351169735, |
| "clip_ratio/low_mean": 9.601536439731717e-05, |
| "clip_ratio/low_min": 9.601536439731717e-05, |
| "clip_ratio/region_mean": 0.0015437977155670524, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 765.0, |
| "completions/max_terminated_length": 765.0, |
| "completions/mean_length": 227.87998962402344, |
| "completions/mean_terminated_length": 227.87998962402344, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.22273644506931306, |
| "epoch": 0.45149253731343286, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009992635808885098, |
| "learning_rate": 5e-05, |
| "loss": 0.0086, |
| "num_tokens": 2730336.0, |
| "reward": 0.6087304353713989, |
| "reward_std": 0.4542500078678131, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.11126953363418579, |
| "rewards/length_penalty/std": 0.058526501059532166, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929081201553345, |
| "sampling/importance_sampling_ratio/min": 0.14870131015777588, |
| "sampling/sampling_logp_difference/max": 1.9058157205581665, |
| "sampling/sampling_logp_difference/mean": 0.0200739074498415, |
| "step": 121, |
| "step_time": 8.137975294841453 |
| }, |
| { |
| "clip_ratio/high_max": 0.001531288563273847, |
| "clip_ratio/high_mean": 0.001531288563273847, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001531288563273847, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 318.0, |
| "completions/max_terminated_length": 318.0, |
| "completions/mean_length": 158.09999084472656, |
| "completions/mean_terminated_length": 158.09999084472656, |
| "completions/min_length": 49.0, |
| "completions/min_terminated_length": 49.0, |
| "entropy": 0.16101946234703063, |
| "epoch": 0.4552238805970149, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010061686858534813, |
| "learning_rate": 5e-05, |
| "loss": -0.0026, |
| "num_tokens": 2740321.0, |
| "reward": 0.8628027439117432, |
| "reward_std": 0.2421693503856659, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.07719726860523224, |
| "rewards/length_penalty/std": 0.036058977246284485, |
| "sampling/importance_sampling_ratio/max": 2.862182140350342, |
| "sampling/importance_sampling_ratio/mean": 0.9939879179000854, |
| "sampling/importance_sampling_ratio/min": 0.17510327696800232, |
| "sampling/sampling_logp_difference/max": 1.7423793077468872, |
| "sampling/sampling_logp_difference/mean": 0.020191431045532227, |
| "step": 122, |
| "step_time": 3.8973080010619015 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016544922895263881, |
| "clip_ratio/high_mean": 0.0016544922895263881, |
| "clip_ratio/low_mean": 0.0001386635034577921, |
| "clip_ratio/low_min": 0.0001386635034577921, |
| "clip_ratio/region_mean": 0.001793155784253031, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 673.0, |
| "completions/mean_length": 264.6399841308594, |
| "completions/mean_terminated_length": 228.24488830566406, |
| "completions/min_length": 59.0, |
| "completions/min_terminated_length": 59.0, |
| "entropy": 0.23895826637744905, |
| "epoch": 0.458955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008277439512312412, |
| "learning_rate": 5e-05, |
| "loss": 0.0044, |
| "num_tokens": 2756553.0, |
| "reward": 0.6307812333106995, |
| "reward_std": 0.4966758191585541, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.12921875715255737, |
| "rewards/length_penalty/std": 0.13625647127628326, |
| "sampling/importance_sampling_ratio/max": 2.4922373294830322, |
| "sampling/importance_sampling_ratio/mean": 0.9901281595230103, |
| "sampling/importance_sampling_ratio/min": 0.0876694768667221, |
| "sampling/sampling_logp_difference/max": 2.4341814517974854, |
| "sampling/sampling_logp_difference/mean": 0.02294175885617733, |
| "step": 123, |
| "step_time": 20.23102223407477 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011977181187830866, |
| "clip_ratio/high_mean": 0.0011977181187830866, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011977181187830866, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 686.0, |
| "completions/max_terminated_length": 686.0, |
| "completions/mean_length": 266.29998779296875, |
| "completions/mean_terminated_length": 266.29998779296875, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.2261672168970108, |
| "epoch": 0.4626865671641791, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008976740762591362, |
| "learning_rate": 5e-05, |
| "loss": 0.004, |
| "num_tokens": 2772218.0, |
| "reward": 0.5299707055091858, |
| "reward_std": 0.5264410376548767, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.13002929091453552, |
| "rewards/length_penalty/std": 0.08299293369054794, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9917804002761841, |
| "sampling/importance_sampling_ratio/min": 0.06805016845464706, |
| "sampling/sampling_logp_difference/max": 2.6875100135803223, |
| "sampling/sampling_logp_difference/mean": 0.02166775055229664, |
| "step": 124, |
| "step_time": 7.64087191876024 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012904302217066287, |
| "clip_ratio/high_mean": 0.0012904302217066287, |
| "clip_ratio/low_mean": 8.795074536465109e-05, |
| "clip_ratio/low_min": 8.795074536465109e-05, |
| "clip_ratio/region_mean": 0.0013783809496089815, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 786.0, |
| "completions/max_terminated_length": 786.0, |
| "completions/mean_length": 231.3199920654297, |
| "completions/mean_terminated_length": 231.3199920654297, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.24359653890132904, |
| "epoch": 0.4664179104477612, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009679988957941532, |
| "learning_rate": 5e-05, |
| "loss": 0.024, |
| "num_tokens": 2786994.0, |
| "reward": 0.6870507597923279, |
| "reward_std": 0.4686749577522278, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.11294922232627869, |
| "rewards/length_penalty/std": 0.07715218514204025, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9902333617210388, |
| "sampling/importance_sampling_ratio/min": 0.007950937375426292, |
| "sampling/sampling_logp_difference/max": 4.834465503692627, |
| "sampling/sampling_logp_difference/mean": 0.02457006461918354, |
| "step": 125, |
| "step_time": 8.776835452066734 |
| }, |
| { |
| "clip_ratio/high_max": 0.00095673818141222, |
| "clip_ratio/high_mean": 0.00095673818141222, |
| "clip_ratio/low_mean": 0.00011344299418851733, |
| "clip_ratio/low_min": 0.00011344299418851733, |
| "clip_ratio/region_mean": 0.0010701811756007374, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 320.0, |
| "completions/max_terminated_length": 320.0, |
| "completions/mean_length": 178.09999084472656, |
| "completions/mean_terminated_length": 178.09999084472656, |
| "completions/min_length": 49.0, |
| "completions/min_terminated_length": 49.0, |
| "entropy": 0.19170283973217012, |
| "epoch": 0.4701492537313433, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008428176864981651, |
| "learning_rate": 5e-05, |
| "loss": 0.0122, |
| "num_tokens": 2798429.0, |
| "reward": 0.5130370855331421, |
| "reward_std": 0.5070653557777405, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.08696289360523224, |
| "rewards/length_penalty/std": 0.03125455230474472, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9930151104927063, |
| "sampling/importance_sampling_ratio/min": 0.10816968977451324, |
| "sampling/sampling_logp_difference/max": 2.2240540981292725, |
| "sampling/sampling_logp_difference/mean": 0.02147875726222992, |
| "step": 126, |
| "step_time": 4.1449216809123755 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006216696347109973, |
| "clip_ratio/high_mean": 0.0006216696347109973, |
| "clip_ratio/low_mean": 0.00010675177909433842, |
| "clip_ratio/low_min": 0.00010675177909433842, |
| "clip_ratio/region_mean": 0.0007284214138053357, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1759.0, |
| "completions/max_terminated_length": 1759.0, |
| "completions/mean_length": 403.1600036621094, |
| "completions/mean_terminated_length": 403.1600036621094, |
| "completions/min_length": 141.0, |
| "completions/min_terminated_length": 141.0, |
| "entropy": 0.23758453130722046, |
| "epoch": 0.47388059701492535, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011349253356456757, |
| "learning_rate": 5e-05, |
| "loss": 0.0219, |
| "num_tokens": 2821917.0, |
| "reward": 0.42314451932907104, |
| "reward_std": 0.5955804586410522, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.19685547053813934, |
| "rewards/length_penalty/std": 0.19482669234275818, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9913946986198425, |
| "sampling/importance_sampling_ratio/min": 0.020159153267741203, |
| "sampling/sampling_logp_difference/max": 3.904096841812134, |
| "sampling/sampling_logp_difference/mean": 0.019592322409152985, |
| "step": 127, |
| "step_time": 18.33419666835107 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008252030820585787, |
| "clip_ratio/high_mean": 0.0008252030820585787, |
| "clip_ratio/low_mean": 0.00014716703444719314, |
| "clip_ratio/low_min": 0.00014716703444719314, |
| "clip_ratio/region_mean": 0.0009723701165057719, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 264.0, |
| "completions/max_terminated_length": 264.0, |
| "completions/mean_length": 156.27999877929688, |
| "completions/mean_terminated_length": 156.27999877929688, |
| "completions/min_length": 67.0, |
| "completions/min_terminated_length": 67.0, |
| "entropy": 0.20920923948287964, |
| "epoch": 0.47761194029850745, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011614021845161915, |
| "learning_rate": 5e-05, |
| "loss": -0.0015, |
| "num_tokens": 2833361.0, |
| "reward": 0.7436913847923279, |
| "reward_std": 0.3907195031642914, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.07630859315395355, |
| "rewards/length_penalty/std": 0.02654467523097992, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9951394200325012, |
| "sampling/importance_sampling_ratio/min": 0.13131150603294373, |
| "sampling/sampling_logp_difference/max": 2.0301828384399414, |
| "sampling/sampling_logp_difference/mean": 0.026422882452607155, |
| "step": 128, |
| "step_time": 3.6463363682851195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010950034542474897, |
| "clip_ratio/high_mean": 0.0010950034542474897, |
| "clip_ratio/low_mean": 0.00024090495426207781, |
| "clip_ratio/low_min": 0.00024090495426207781, |
| "clip_ratio/region_mean": 0.001335908385226503, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 533.0, |
| "completions/max_terminated_length": 533.0, |
| "completions/mean_length": 194.0, |
| "completions/mean_terminated_length": 194.0, |
| "completions/min_length": 80.0, |
| "completions/min_terminated_length": 80.0, |
| "entropy": 0.22849051356315614, |
| "epoch": 0.48134328358208955, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010003703646361828, |
| "learning_rate": 5e-05, |
| "loss": 0.0003, |
| "num_tokens": 2846081.0, |
| "reward": 0.3052734434604645, |
| "reward_std": 0.49904265999794006, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.0947265625, |
| "rewards/length_penalty/std": 0.06281977146863937, |
| "sampling/importance_sampling_ratio/max": 2.360441207885742, |
| "sampling/importance_sampling_ratio/mean": 0.9916234612464905, |
| "sampling/importance_sampling_ratio/min": 0.07548636943101883, |
| "sampling/sampling_logp_difference/max": 2.583803176879883, |
| "sampling/sampling_logp_difference/mean": 0.025569908320903778, |
| "step": 129, |
| "step_time": 6.192054107086733 |
| }, |
| { |
| "clip_ratio/high_max": 0.0019061979372054338, |
| "clip_ratio/high_mean": 0.0019061979372054338, |
| "clip_ratio/low_mean": 0.00011142061557620764, |
| "clip_ratio/low_min": 0.00011142061557620764, |
| "clip_ratio/region_mean": 0.002017618576064706, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 862.0, |
| "completions/max_terminated_length": 862.0, |
| "completions/mean_length": 172.3199920654297, |
| "completions/mean_terminated_length": 172.3199920654297, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.2615941911935806, |
| "epoch": 0.48507462686567165, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010521075688302517, |
| "learning_rate": 5e-05, |
| "loss": 0.0034, |
| "num_tokens": 2857277.0, |
| "reward": 0.8158593773841858, |
| "reward_std": 0.34243932366371155, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304574966430664, |
| "rewards/length_penalty/mean": -0.08414062857627869, |
| "rewards/length_penalty/std": 0.07409535348415375, |
| "sampling/importance_sampling_ratio/max": 2.831399917602539, |
| "sampling/importance_sampling_ratio/mean": 0.9886124134063721, |
| "sampling/importance_sampling_ratio/min": 0.20766989886760712, |
| "sampling/sampling_logp_difference/max": 1.571805477142334, |
| "sampling/sampling_logp_difference/mean": 0.02765255607664585, |
| "step": 130, |
| "step_time": 8.862665467895567 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006989035930018872, |
| "clip_ratio/high_mean": 0.0006989035930018872, |
| "clip_ratio/low_mean": 0.000265103334095329, |
| "clip_ratio/low_min": 0.000265103334095329, |
| "clip_ratio/region_mean": 0.0009640069212764502, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1877.0, |
| "completions/mean_length": 522.7799682617188, |
| "completions/mean_terminated_length": 232.26190185546875, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.3432747036218643, |
| "epoch": 0.48880597014925375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011200115084648132, |
| "learning_rate": 5e-05, |
| "loss": 0.0352, |
| "num_tokens": 2886536.0, |
| "reward": 0.5247363448143005, |
| "reward_std": 0.7512145638465881, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.25526368618011475, |
| "rewards/length_penalty/std": 0.3582891523838043, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.986057698726654, |
| "sampling/importance_sampling_ratio/min": 0.25044798851013184, |
| "sampling/sampling_logp_difference/max": 1.4710482358932495, |
| "sampling/sampling_logp_difference/mean": 0.026101168245077133, |
| "step": 131, |
| "step_time": 22.4152356677223 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004852790618315339, |
| "clip_ratio/high_mean": 0.0004852790618315339, |
| "clip_ratio/low_mean": 0.00024426057934761046, |
| "clip_ratio/low_min": 0.00024426057934761046, |
| "clip_ratio/region_mean": 0.0007295396411791444, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 243.0, |
| "completions/max_terminated_length": 243.0, |
| "completions/mean_length": 166.1199951171875, |
| "completions/mean_terminated_length": 166.1199951171875, |
| "completions/min_length": 115.0, |
| "completions/min_terminated_length": 115.0, |
| "entropy": 0.15842755734920502, |
| "epoch": 0.4925373134328358, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00819218996912241, |
| "learning_rate": 5e-05, |
| "loss": 0.0093, |
| "num_tokens": 2897642.0, |
| "reward": 0.71888667345047, |
| "reward_std": 0.41740432381629944, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.08111327886581421, |
| "rewards/length_penalty/std": 0.01795812137424946, |
| "sampling/importance_sampling_ratio/max": 2.10459303855896, |
| "sampling/importance_sampling_ratio/mean": 0.9947901964187622, |
| "sampling/importance_sampling_ratio/min": 0.20642346143722534, |
| "sampling/sampling_logp_difference/max": 1.5778255462646484, |
| "sampling/sampling_logp_difference/mean": 0.02011234685778618, |
| "step": 132, |
| "step_time": 3.3312100879848003 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012222370714880526, |
| "clip_ratio/high_mean": 0.0012222370714880526, |
| "clip_ratio/low_mean": 0.00019110393477603794, |
| "clip_ratio/low_min": 0.00019110393477603794, |
| "clip_ratio/region_mean": 0.0014133410179056228, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 814.0, |
| "completions/max_terminated_length": 814.0, |
| "completions/mean_length": 198.86000061035156, |
| "completions/mean_terminated_length": 198.86000061035156, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.24506987929344176, |
| "epoch": 0.4962686567164179, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01176573894917965, |
| "learning_rate": 5e-05, |
| "loss": 0.0218, |
| "num_tokens": 2910595.0, |
| "reward": 0.8429003953933716, |
| "reward_std": 0.2683672606945038, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.09709960967302322, |
| "rewards/length_penalty/std": 0.05587049946188927, |
| "sampling/importance_sampling_ratio/max": 2.8337061405181885, |
| "sampling/importance_sampling_ratio/mean": 0.9913595914840698, |
| "sampling/importance_sampling_ratio/min": 0.08304513990879059, |
| "sampling/sampling_logp_difference/max": 2.488370895385742, |
| "sampling/sampling_logp_difference/mean": 0.027137257158756256, |
| "step": 133, |
| "step_time": 8.549008164089173 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005050760868471116, |
| "clip_ratio/high_mean": 0.0005050760868471116, |
| "clip_ratio/low_mean": 7.791196112520993e-05, |
| "clip_ratio/low_min": 7.791196112520993e-05, |
| "clip_ratio/region_mean": 0.0005829880421515555, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 591.0, |
| "completions/max_terminated_length": 591.0, |
| "completions/mean_length": 231.87998962402344, |
| "completions/mean_terminated_length": 231.87998962402344, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.20832602977752684, |
| "epoch": 0.5, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007731511723250151, |
| "learning_rate": 5e-05, |
| "loss": -0.0079, |
| "num_tokens": 2924439.0, |
| "reward": 0.5067773461341858, |
| "reward_std": 0.4746435284614563, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.11322265863418579, |
| "rewards/length_penalty/std": 0.053937532007694244, |
| "sampling/importance_sampling_ratio/max": 2.3509459495544434, |
| "sampling/importance_sampling_ratio/mean": 0.9926426410675049, |
| "sampling/importance_sampling_ratio/min": 0.012443562038242817, |
| "sampling/sampling_logp_difference/max": 4.386551856994629, |
| "sampling/sampling_logp_difference/mean": 0.023071065545082092, |
| "step": 134, |
| "step_time": 6.65445177606307 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013679137162398547, |
| "clip_ratio/high_mean": 0.0013679137162398547, |
| "clip_ratio/low_mean": 0.0001430522301234305, |
| "clip_ratio/low_min": 0.0001430522301234305, |
| "clip_ratio/region_mean": 0.0015109659521840512, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1587.0, |
| "completions/mean_length": 359.8999938964844, |
| "completions/mean_terminated_length": 289.5625, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.30269221067428587, |
| "epoch": 0.503731343283582, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011205162853002548, |
| "learning_rate": 5e-05, |
| "loss": 0.0081, |
| "num_tokens": 2945054.0, |
| "reward": 0.5842675566673279, |
| "reward_std": 0.6361220479011536, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.17573241889476776, |
| "rewards/length_penalty/std": 0.24155470728874207, |
| "sampling/importance_sampling_ratio/max": 2.0634121894836426, |
| "sampling/importance_sampling_ratio/mean": 0.9888497591018677, |
| "sampling/importance_sampling_ratio/min": 0.2574204206466675, |
| "sampling/sampling_logp_difference/max": 1.3570445775985718, |
| "sampling/sampling_logp_difference/mean": 0.024053573608398438, |
| "step": 135, |
| "step_time": 21.20175286172889 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006372028379701078, |
| "clip_ratio/high_mean": 0.0006372028379701078, |
| "clip_ratio/low_mean": 0.00027050451608374716, |
| "clip_ratio/low_min": 0.00027050451608374716, |
| "clip_ratio/region_mean": 0.0009077073540538549, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 288.0, |
| "completions/max_terminated_length": 288.0, |
| "completions/mean_length": 159.1999969482422, |
| "completions/mean_terminated_length": 159.1999969482422, |
| "completions/min_length": 82.0, |
| "completions/min_terminated_length": 82.0, |
| "entropy": 0.1909633070230484, |
| "epoch": 0.5074626865671642, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008253993466496468, |
| "learning_rate": 5e-05, |
| "loss": 0.0009, |
| "num_tokens": 2955834.0, |
| "reward": 0.5822656154632568, |
| "reward_std": 0.48381224274635315, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.07773437350988388, |
| "rewards/length_penalty/std": 0.023210110142827034, |
| "sampling/importance_sampling_ratio/max": 2.899434804916382, |
| "sampling/importance_sampling_ratio/mean": 0.9926809072494507, |
| "sampling/importance_sampling_ratio/min": 0.19921831786632538, |
| "sampling/sampling_logp_difference/max": 1.613353967666626, |
| "sampling/sampling_logp_difference/mean": 0.026710735633969307, |
| "step": 136, |
| "step_time": 3.7305929258000106 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009868979221209883, |
| "clip_ratio/high_mean": 0.0009868979221209883, |
| "clip_ratio/low_mean": 0.00014139271806925535, |
| "clip_ratio/low_min": 0.00014139271806925535, |
| "clip_ratio/region_mean": 0.001128290651831776, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1179.0, |
| "completions/max_terminated_length": 1179.0, |
| "completions/mean_length": 203.51998901367188, |
| "completions/mean_terminated_length": 203.51998901367188, |
| "completions/min_length": 49.0, |
| "completions/min_terminated_length": 49.0, |
| "entropy": 0.16949701309204102, |
| "epoch": 0.5111940298507462, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00999395176768303, |
| "learning_rate": 5e-05, |
| "loss": 0.0331, |
| "num_tokens": 2969090.0, |
| "reward": 0.5006250143051147, |
| "reward_std": 0.5276067852973938, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.09937500208616257, |
| "rewards/length_penalty/std": 0.0836191177368164, |
| "sampling/importance_sampling_ratio/max": 2.5715415477752686, |
| "sampling/importance_sampling_ratio/mean": 0.9937620162963867, |
| "sampling/importance_sampling_ratio/min": 0.16146661341190338, |
| "sampling/sampling_logp_difference/max": 1.823456883430481, |
| "sampling/sampling_logp_difference/mean": 0.021660471335053444, |
| "step": 137, |
| "step_time": 11.875488157849759 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013344243518076837, |
| "clip_ratio/high_mean": 0.0013344243518076837, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0013344243518076837, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 595.0, |
| "completions/max_terminated_length": 595.0, |
| "completions/mean_length": 182.22000122070312, |
| "completions/mean_terminated_length": 182.22000122070312, |
| "completions/min_length": 56.0, |
| "completions/min_terminated_length": 56.0, |
| "entropy": 0.2002437561750412, |
| "epoch": 0.5149253731343284, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007798939011991024, |
| "learning_rate": 5e-05, |
| "loss": 0.001, |
| "num_tokens": 2980551.0, |
| "reward": 0.47102537751197815, |
| "reward_std": 0.5201479196548462, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.08897460997104645, |
| "rewards/length_penalty/std": 0.046709589660167694, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927992820739746, |
| "sampling/importance_sampling_ratio/min": 0.25913453102111816, |
| "sampling/sampling_logp_difference/max": 1.3504079580307007, |
| "sampling/sampling_logp_difference/mean": 0.02293580397963524, |
| "step": 138, |
| "step_time": 6.858032171847299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007743010413832962, |
| "clip_ratio/high_mean": 0.0007743010413832962, |
| "clip_ratio/low_mean": 0.0001226993859745562, |
| "clip_ratio/low_min": 0.0001226993859745562, |
| "clip_ratio/region_mean": 0.0008970004273578525, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 907.0, |
| "completions/max_terminated_length": 907.0, |
| "completions/mean_length": 240.739990234375, |
| "completions/mean_terminated_length": 240.739990234375, |
| "completions/min_length": 65.0, |
| "completions/min_terminated_length": 65.0, |
| "entropy": 0.3281055152416229, |
| "epoch": 0.5186567164179104, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016581956297159195, |
| "learning_rate": 5e-05, |
| "loss": -0.0298, |
| "num_tokens": 2995598.0, |
| "reward": 0.7224511504173279, |
| "reward_std": 0.42162489891052246, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.11754883080720901, |
| "rewards/length_penalty/std": 0.1017332375049591, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9883118867874146, |
| "sampling/importance_sampling_ratio/min": 0.004260603804141283, |
| "sampling/sampling_logp_difference/max": 5.458344459533691, |
| "sampling/sampling_logp_difference/mean": 0.028877798467874527, |
| "step": 139, |
| "step_time": 9.61821757024154 |
| }, |
| { |
| "clip_ratio/high_max": 0.00094691306585446, |
| "clip_ratio/high_mean": 0.00094691306585446, |
| "clip_ratio/low_mean": 0.00011179429711773991, |
| "clip_ratio/low_min": 0.00011179429711773991, |
| "clip_ratio/region_mean": 0.0010587073629722, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 272.0, |
| "completions/max_terminated_length": 272.0, |
| "completions/mean_length": 172.59999084472656, |
| "completions/mean_terminated_length": 172.59999084472656, |
| "completions/min_length": 53.0, |
| "completions/min_terminated_length": 53.0, |
| "entropy": 0.18536655306816102, |
| "epoch": 0.5223880597014925, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006095483899116516, |
| "learning_rate": 5e-05, |
| "loss": -0.0007, |
| "num_tokens": 3007278.0, |
| "reward": 0.8957226276397705, |
| "reward_std": 0.13739050924777985, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.08427734673023224, |
| "rewards/length_penalty/std": 0.027954401448369026, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9943463802337646, |
| "sampling/importance_sampling_ratio/min": 0.28493720293045044, |
| "sampling/sampling_logp_difference/max": 1.5161831378936768, |
| "sampling/sampling_logp_difference/mean": 0.02388784848153591, |
| "step": 140, |
| "step_time": 3.617985596181825 |
| }, |
| { |
| "clip_ratio/high_max": 0.001352481311187148, |
| "clip_ratio/high_mean": 0.001352481311187148, |
| "clip_ratio/low_mean": 8.156606927514076e-05, |
| "clip_ratio/low_min": 8.156606927514076e-05, |
| "clip_ratio/region_mean": 0.0014340473804622888, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1019.0, |
| "completions/max_terminated_length": 1019.0, |
| "completions/mean_length": 217.05999755859375, |
| "completions/mean_terminated_length": 217.05999755859375, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.2602251559495926, |
| "epoch": 0.5261194029850746, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01131103839725256, |
| "learning_rate": 5e-05, |
| "loss": 0.016, |
| "num_tokens": 3021121.0, |
| "reward": 0.6340136528015137, |
| "reward_std": 0.46149131655693054, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.1059863269329071, |
| "rewards/length_penalty/std": 0.08599921315908432, |
| "sampling/importance_sampling_ratio/max": 1.9917515516281128, |
| "sampling/importance_sampling_ratio/mean": 0.9909331798553467, |
| "sampling/importance_sampling_ratio/min": 0.044206660240888596, |
| "sampling/sampling_logp_difference/max": 3.118879795074463, |
| "sampling/sampling_logp_difference/mean": 0.02522987872362137, |
| "step": 141, |
| "step_time": 10.754487411119044 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007487521739676595, |
| "clip_ratio/high_mean": 0.0007487521739676595, |
| "clip_ratio/low_mean": 0.00038297356804832815, |
| "clip_ratio/low_min": 0.00038297356804832815, |
| "clip_ratio/region_mean": 0.0011317257303744555, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1362.0, |
| "completions/max_terminated_length": 1362.0, |
| "completions/mean_length": 208.67999267578125, |
| "completions/mean_terminated_length": 208.67999267578125, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.27973111867904665, |
| "epoch": 0.5298507462686567, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00975009799003601, |
| "learning_rate": 5e-05, |
| "loss": 0.0074, |
| "num_tokens": 3034865.0, |
| "reward": 0.55810546875, |
| "reward_std": 0.5199998617172241, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.10189452767372131, |
| "rewards/length_penalty/std": 0.1248294860124588, |
| "sampling/importance_sampling_ratio/max": 2.914943218231201, |
| "sampling/importance_sampling_ratio/mean": 0.9886429309844971, |
| "sampling/importance_sampling_ratio/min": 0.1923140287399292, |
| "sampling/sampling_logp_difference/max": 1.6486256122589111, |
| "sampling/sampling_logp_difference/mean": 0.02921033836901188, |
| "step": 142, |
| "step_time": 13.86586976586841 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007583755766972899, |
| "clip_ratio/high_mean": 0.0007583755766972899, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007583755766972899, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 186.0, |
| "completions/max_terminated_length": 186.0, |
| "completions/mean_length": 104.0199966430664, |
| "completions/mean_terminated_length": 104.0199966430664, |
| "completions/min_length": 56.0, |
| "completions/min_terminated_length": 56.0, |
| "entropy": 0.23494349122047425, |
| "epoch": 0.5335820895522388, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008589287288486958, |
| "learning_rate": 5e-05, |
| "loss": 0.007, |
| "num_tokens": 3042206.0, |
| "reward": 0.7492089867591858, |
| "reward_std": 0.4037863314151764, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.05079101398587227, |
| "rewards/length_penalty/std": 0.017894024029374123, |
| "sampling/importance_sampling_ratio/max": 2.440413236618042, |
| "sampling/importance_sampling_ratio/mean": 0.989648163318634, |
| "sampling/importance_sampling_ratio/min": 0.3682730495929718, |
| "sampling/sampling_logp_difference/max": 0.9989306330680847, |
| "sampling/sampling_logp_difference/mean": 0.03275414556264877, |
| "step": 143, |
| "step_time": 2.707681803731248 |
| }, |
| { |
| "clip_ratio/high_max": 0.00039639030583202837, |
| "clip_ratio/high_mean": 0.00039639030583202837, |
| "clip_ratio/low_mean": 0.00012492192909121513, |
| "clip_ratio/low_min": 0.00012492192909121513, |
| "clip_ratio/region_mean": 0.0005213122349232435, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 281.0, |
| "completions/max_terminated_length": 281.0, |
| "completions/mean_length": 148.05999755859375, |
| "completions/mean_terminated_length": 148.05999755859375, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.1941971868276596, |
| "epoch": 0.5373134328358209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008025547489523888, |
| "learning_rate": 5e-05, |
| "loss": -0.0018, |
| "num_tokens": 3052429.0, |
| "reward": 0.5477050542831421, |
| "reward_std": 0.48588624596595764, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143346309662, |
| "rewards/length_penalty/mean": -0.0722949206829071, |
| "rewards/length_penalty/std": 0.029356837272644043, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9948816299438477, |
| "sampling/importance_sampling_ratio/min": 0.1723131537437439, |
| "sampling/sampling_logp_difference/max": 1.7584418058395386, |
| "sampling/sampling_logp_difference/mean": 0.027595320716500282, |
| "step": 144, |
| "step_time": 3.6917892990168184 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007910300511866808, |
| "clip_ratio/high_mean": 0.0007910300511866808, |
| "clip_ratio/low_mean": 5.158627755008638e-05, |
| "clip_ratio/low_min": 5.158627755008638e-05, |
| "clip_ratio/region_mean": 0.0008426163345575333, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1072.0, |
| "completions/max_terminated_length": 1072.0, |
| "completions/mean_length": 378.79998779296875, |
| "completions/mean_terminated_length": 378.79998779296875, |
| "completions/min_length": 146.0, |
| "completions/min_terminated_length": 146.0, |
| "entropy": 0.24506783187389375, |
| "epoch": 0.5410447761194029, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010319409891963005, |
| "learning_rate": 5e-05, |
| "loss": 0.0037, |
| "num_tokens": 3074779.0, |
| "reward": 0.5150390267372131, |
| "reward_std": 0.4760143458843231, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.18496093153953552, |
| "rewards/length_penalty/std": 0.10473135113716125, |
| "sampling/importance_sampling_ratio/max": 2.6398367881774902, |
| "sampling/importance_sampling_ratio/mean": 0.9905233979225159, |
| "sampling/importance_sampling_ratio/min": 0.004154582507908344, |
| "sampling/sampling_logp_difference/max": 5.483543395996094, |
| "sampling/sampling_logp_difference/mean": 0.022855687886476517, |
| "step": 145, |
| "step_time": 12.111670856829733 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012341871159151196, |
| "clip_ratio/high_mean": 0.0012341871159151196, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0012341871159151196, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 357.0, |
| "completions/max_terminated_length": 357.0, |
| "completions/mean_length": 144.86000061035156, |
| "completions/mean_terminated_length": 144.86000061035156, |
| "completions/min_length": 75.0, |
| "completions/min_terminated_length": 75.0, |
| "entropy": 0.20047489702701568, |
| "epoch": 0.5447761194029851, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010429944843053818, |
| "learning_rate": 5e-05, |
| "loss": 0.0128, |
| "num_tokens": 3084632.0, |
| "reward": 0.7292675375938416, |
| "reward_std": 0.3980523645877838, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.07073242217302322, |
| "rewards/length_penalty/std": 0.031494587659835815, |
| "sampling/importance_sampling_ratio/max": 2.825479030609131, |
| "sampling/importance_sampling_ratio/mean": 0.9931192994117737, |
| "sampling/importance_sampling_ratio/min": 0.37200209498405457, |
| "sampling/sampling_logp_difference/max": 1.0386779308319092, |
| "sampling/sampling_logp_difference/mean": 0.027243606746196747, |
| "step": 146, |
| "step_time": 4.238418721826747 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016277542687021195, |
| "clip_ratio/high_mean": 0.0016277542687021195, |
| "clip_ratio/low_mean": 0.00031077241292223336, |
| "clip_ratio/low_min": 0.00031077241292223336, |
| "clip_ratio/region_mean": 0.0019385267049074173, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 253.0, |
| "completions/max_terminated_length": 253.0, |
| "completions/mean_length": 105.15999603271484, |
| "completions/mean_terminated_length": 105.15999603271484, |
| "completions/min_length": 46.0, |
| "completions/min_terminated_length": 46.0, |
| "entropy": 0.22340488731861113, |
| "epoch": 0.5485074626865671, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008171666413545609, |
| "learning_rate": 5e-05, |
| "loss": 0.0025, |
| "num_tokens": 3092000.0, |
| "reward": 0.8486523032188416, |
| "reward_std": 0.3088226914405823, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.30304577946662903, |
| "rewards/length_penalty/mean": -0.05134765803813934, |
| "rewards/length_penalty/std": 0.024106472730636597, |
| "sampling/importance_sampling_ratio/max": 2.0590014457702637, |
| "sampling/importance_sampling_ratio/mean": 0.9904125332832336, |
| "sampling/importance_sampling_ratio/min": 0.13179636001586914, |
| "sampling/sampling_logp_difference/max": 2.0264973640441895, |
| "sampling/sampling_logp_difference/mean": 0.0369202122092247, |
| "step": 147, |
| "step_time": 3.1389055219478905 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009367710794322192, |
| "clip_ratio/high_mean": 0.0009367710794322192, |
| "clip_ratio/low_mean": 0.00044899251079186795, |
| "clip_ratio/low_min": 0.00044899251079186795, |
| "clip_ratio/region_mean": 0.0013857635902240872, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 384.0, |
| "completions/max_terminated_length": 384.0, |
| "completions/mean_length": 125.0999984741211, |
| "completions/mean_terminated_length": 125.0999984741211, |
| "completions/min_length": 41.0, |
| "completions/min_terminated_length": 41.0, |
| "entropy": 0.22060433626174927, |
| "epoch": 0.5522388059701493, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007559936493635178, |
| "learning_rate": 5e-05, |
| "loss": -0.001, |
| "num_tokens": 3100565.0, |
| "reward": 0.2989160120487213, |
| "reward_std": 0.5032209753990173, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.06108398362994194, |
| "rewards/length_penalty/std": 0.03567451238632202, |
| "sampling/importance_sampling_ratio/max": 2.571481943130493, |
| "sampling/importance_sampling_ratio/mean": 0.9903957843780518, |
| "sampling/importance_sampling_ratio/min": 0.22772671282291412, |
| "sampling/sampling_logp_difference/max": 1.4796090126037598, |
| "sampling/sampling_logp_difference/mean": 0.03094572387635708, |
| "step": 148, |
| "step_time": 4.429547405568883 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008332037832587957, |
| "clip_ratio/high_mean": 0.0008332037832587957, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008332037832587957, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 441.0, |
| "completions/max_terminated_length": 441.0, |
| "completions/mean_length": 167.44000244140625, |
| "completions/mean_terminated_length": 167.44000244140625, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.23512509763240813, |
| "epoch": 0.5559701492537313, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009715686552226543, |
| "learning_rate": 5e-05, |
| "loss": -0.0043, |
| "num_tokens": 3111857.0, |
| "reward": 0.4382421672344208, |
| "reward_std": 0.49810686707496643, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.0817578136920929, |
| "rewards/length_penalty/std": 0.0382026769220829, |
| "sampling/importance_sampling_ratio/max": 2.1647255420684814, |
| "sampling/importance_sampling_ratio/mean": 0.989840567111969, |
| "sampling/importance_sampling_ratio/min": 0.22186876833438873, |
| "sampling/sampling_logp_difference/max": 1.5056692361831665, |
| "sampling/sampling_logp_difference/mean": 0.027890125289559364, |
| "step": 149, |
| "step_time": 5.110821545822546 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009919501142576336, |
| "clip_ratio/high_mean": 0.0009919501142576336, |
| "clip_ratio/low_mean": 7.98722030594945e-05, |
| "clip_ratio/low_min": 7.98722030594945e-05, |
| "clip_ratio/region_mean": 0.0010718223173171281, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1053.0, |
| "completions/max_terminated_length": 1053.0, |
| "completions/mean_length": 218.89999389648438, |
| "completions/mean_terminated_length": 218.89999389648438, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.21045926809310914, |
| "epoch": 0.5597014925373134, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017433729022741318, |
| "learning_rate": 5e-05, |
| "loss": 0.0439, |
| "num_tokens": 3126652.0, |
| "reward": 0.4731152355670929, |
| "reward_std": 0.5139723420143127, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.10688476264476776, |
| "rewards/length_penalty/std": 0.0894477590918541, |
| "sampling/importance_sampling_ratio/max": 2.910443067550659, |
| "sampling/importance_sampling_ratio/mean": 0.9920240044593811, |
| "sampling/importance_sampling_ratio/min": 0.3026241362094879, |
| "sampling/sampling_logp_difference/max": 1.1952637434005737, |
| "sampling/sampling_logp_difference/mean": 0.02535368502140045, |
| "step": 150, |
| "step_time": 11.092629134189337 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 200, |
| "num_input_tokens_seen": 3126652, |
| "num_train_epochs": 1, |
| "save_steps": 50, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|