| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.373134328358209, |
| "eval_steps": 500, |
| "global_step": 100, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 567.0, |
| "completions/max_terminated_length": 567.0, |
| "completions/mean_length": 259.2799987792969, |
| "completions/mean_terminated_length": 259.2799987792969, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.1668988436460495, |
| "epoch": 0.0037313432835820895, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006947502493858337, |
| "learning_rate": 0.0, |
| "loss": 0.0044, |
| "num_tokens": 15194.0, |
| "reward": 0.7933984398841858, |
| "reward_std": 0.29082080721855164, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.12660156190395355, |
| "rewards/length_penalty/std": 0.04607566446065903, |
| "sampling/importance_sampling_ratio/max": 1.3554484844207764, |
| "sampling/importance_sampling_ratio/mean": 0.9939242601394653, |
| "sampling/importance_sampling_ratio/min": 0.6691522598266602, |
| "sampling/sampling_logp_difference/max": 0.40174365043640137, |
| "sampling/sampling_logp_difference/mean": 0.011276595294475555, |
| "step": 1, |
| "step_time": 6.9493025781121105 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1107.0, |
| "completions/max_terminated_length": 1107.0, |
| "completions/mean_length": 412.0, |
| "completions/mean_terminated_length": 412.0, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.1244592621922493, |
| "epoch": 0.007462686567164179, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007026078645139933, |
| "learning_rate": 5e-06, |
| "loss": 0.0366, |
| "num_tokens": 38684.0, |
| "reward": 0.3988281190395355, |
| "reward_std": 0.5786729454994202, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.201171875, |
| "rewards/length_penalty/std": 0.12524312734603882, |
| "sampling/importance_sampling_ratio/max": 1.4278866052627563, |
| "sampling/importance_sampling_ratio/mean": 0.9958950281143188, |
| "sampling/importance_sampling_ratio/min": 0.6846740245819092, |
| "sampling/sampling_logp_difference/max": 0.37881243228912354, |
| "sampling/sampling_logp_difference/mean": 0.008306741714477539, |
| "step": 2, |
| "step_time": 12.598701566690579 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007442941830959171, |
| "clip_ratio/high_mean": 0.0007442941830959171, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007442941830959171, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 672.0, |
| "completions/max_terminated_length": 672.0, |
| "completions/mean_length": 306.5799865722656, |
| "completions/mean_terminated_length": 306.5799865722656, |
| "completions/min_length": 188.0, |
| "completions/min_terminated_length": 188.0, |
| "entropy": 0.16790179610252381, |
| "epoch": 0.011194029850746268, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008647753857076168, |
| "learning_rate": 1e-05, |
| "loss": 0.0019, |
| "num_tokens": 56933.0, |
| "reward": 0.4303027391433716, |
| "reward_std": 0.4748076796531677, |
| "rewards/correctness/mean": 0.5799999833106995, |
| "rewards/correctness/std": 0.49856939911842346, |
| "rewards/length_penalty/mean": -0.14969725906848907, |
| "rewards/length_penalty/std": 0.05218725651502609, |
| "sampling/importance_sampling_ratio/max": 1.413362741470337, |
| "sampling/importance_sampling_ratio/mean": 0.993857741355896, |
| "sampling/importance_sampling_ratio/min": 0.6936461925506592, |
| "sampling/sampling_logp_difference/max": 0.36579322814941406, |
| "sampling/sampling_logp_difference/mean": 0.011682353913784027, |
| "step": 3, |
| "step_time": 8.017057088203728 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006428720662370324, |
| "clip_ratio/high_mean": 0.0006428720662370324, |
| "clip_ratio/low_mean": 4.096681659575552e-05, |
| "clip_ratio/low_min": 4.096681659575552e-05, |
| "clip_ratio/region_mean": 0.000683838885743171, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1016.0, |
| "completions/max_terminated_length": 1016.0, |
| "completions/mean_length": 445.1600036621094, |
| "completions/mean_terminated_length": 445.1600036621094, |
| "completions/min_length": 205.0, |
| "completions/min_terminated_length": 205.0, |
| "entropy": 0.12619452476501464, |
| "epoch": 0.014925373134328358, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007980461232364178, |
| "learning_rate": 1.5e-05, |
| "loss": 0.0268, |
| "num_tokens": 82271.0, |
| "reward": 0.5826367139816284, |
| "reward_std": 0.5049587488174438, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.21736328303813934, |
| "rewards/length_penalty/std": 0.11474549770355225, |
| "sampling/importance_sampling_ratio/max": 1.430497407913208, |
| "sampling/importance_sampling_ratio/mean": 0.9957796931266785, |
| "sampling/importance_sampling_ratio/min": 0.7457529306411743, |
| "sampling/sampling_logp_difference/max": 0.35802221298217773, |
| "sampling/sampling_logp_difference/mean": 0.00873645767569542, |
| "step": 4, |
| "step_time": 12.020404138602316 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011642194353044034, |
| "clip_ratio/high_mean": 0.0011642194353044034, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011642194353044034, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 601.0, |
| "completions/max_terminated_length": 601.0, |
| "completions/mean_length": 290.8800048828125, |
| "completions/mean_terminated_length": 290.8800048828125, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.1662202298641205, |
| "epoch": 0.018656716417910446, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005730130709707737, |
| "learning_rate": 2e-05, |
| "loss": 0.0008, |
| "num_tokens": 99835.0, |
| "reward": 0.8379687070846558, |
| "reward_std": 0.15626150369644165, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1420312523841858, |
| "rewards/length_penalty/std": 0.06337719410657883, |
| "sampling/importance_sampling_ratio/max": 1.3366026878356934, |
| "sampling/importance_sampling_ratio/mean": 0.9940405488014221, |
| "sampling/importance_sampling_ratio/min": 0.7202601432800293, |
| "sampling/sampling_logp_difference/max": 0.3281428813934326, |
| "sampling/sampling_logp_difference/mean": 0.011546244844794273, |
| "step": 5, |
| "step_time": 7.627342459047213 |
| }, |
| { |
| "clip_ratio/high_max": 0.000865393309504725, |
| "clip_ratio/high_mean": 0.000865393309504725, |
| "clip_ratio/low_mean": 4.660918202716857e-05, |
| "clip_ratio/low_min": 4.660918202716857e-05, |
| "clip_ratio/region_mean": 0.0009120024915318936, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1064.0, |
| "completions/max_terminated_length": 1064.0, |
| "completions/mean_length": 485.53997802734375, |
| "completions/mean_terminated_length": 485.53997802734375, |
| "completions/min_length": 220.0, |
| "completions/min_terminated_length": 220.0, |
| "entropy": 0.14179427921772003, |
| "epoch": 0.022388059701492536, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008097657933831215, |
| "learning_rate": 2.5e-05, |
| "loss": -0.0103, |
| "num_tokens": 127112.0, |
| "reward": 0.7229198813438416, |
| "reward_std": 0.240312397480011, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.23708008229732513, |
| "rewards/length_penalty/std": 0.11706869304180145, |
| "sampling/importance_sampling_ratio/max": 1.4403629302978516, |
| "sampling/importance_sampling_ratio/mean": 0.9951878190040588, |
| "sampling/importance_sampling_ratio/min": 0.6609635949134827, |
| "sampling/sampling_logp_difference/max": 0.41405653953552246, |
| "sampling/sampling_logp_difference/mean": 0.009446932002902031, |
| "step": 6, |
| "step_time": 12.106442798860371 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007034160080365837, |
| "clip_ratio/high_mean": 0.0007034160080365837, |
| "clip_ratio/low_mean": 5.22593007190153e-05, |
| "clip_ratio/low_min": 5.22593007190153e-05, |
| "clip_ratio/region_mean": 0.0007556752883829176, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2037.0, |
| "completions/mean_length": 686.5599975585938, |
| "completions/mean_terminated_length": 599.6595458984375, |
| "completions/min_length": 212.0, |
| "completions/min_terminated_length": 212.0, |
| "entropy": 0.2536372452974319, |
| "epoch": 0.026119402985074626, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00851096399128437, |
| "learning_rate": 3e-05, |
| "loss": 0.0075, |
| "num_tokens": 165200.0, |
| "reward": 0.36476561427116394, |
| "reward_std": 0.7040084004402161, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.3352343738079071, |
| "rewards/length_penalty/std": 0.2852962613105774, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9906913638114929, |
| "sampling/importance_sampling_ratio/min": 0.676866352558136, |
| "sampling/sampling_logp_difference/max": 1.6592447757720947, |
| "sampling/sampling_logp_difference/mean": 0.016186492517590523, |
| "step": 7, |
| "step_time": 22.643113143742085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004955741052981466, |
| "clip_ratio/high_mean": 0.0004955741052981466, |
| "clip_ratio/low_mean": 0.00013108916173223406, |
| "clip_ratio/low_min": 0.00013108916173223406, |
| "clip_ratio/region_mean": 0.0006266632699407637, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1593.0, |
| "completions/mean_length": 585.8999633789062, |
| "completions/mean_terminated_length": 492.574462890625, |
| "completions/min_length": 210.0, |
| "completions/min_terminated_length": 210.0, |
| "entropy": 0.22597861886024476, |
| "epoch": 0.029850746268656716, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015066474676132202, |
| "learning_rate": 3.5e-05, |
| "loss": 0.1228, |
| "num_tokens": 198345.0, |
| "reward": 0.19391600787639618, |
| "reward_std": 0.6122970581054688, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.28608399629592896, |
| "rewards/length_penalty/std": 0.23632247745990753, |
| "sampling/importance_sampling_ratio/max": 1.5942074060440063, |
| "sampling/importance_sampling_ratio/mean": 0.992152214050293, |
| "sampling/importance_sampling_ratio/min": 0.6727555394172668, |
| "sampling/sampling_logp_difference/max": 0.4663766622543335, |
| "sampling/sampling_logp_difference/mean": 0.014843679964542389, |
| "step": 8, |
| "step_time": 22.47531992616132 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006101833190768957, |
| "clip_ratio/high_mean": 0.0006101833190768957, |
| "clip_ratio/low_mean": 5.7159189600497486e-05, |
| "clip_ratio/low_min": 5.7159189600497486e-05, |
| "clip_ratio/region_mean": 0.0006673425086773932, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 587.0, |
| "completions/max_terminated_length": 587.0, |
| "completions/mean_length": 336.1399841308594, |
| "completions/mean_terminated_length": 336.1399841308594, |
| "completions/min_length": 216.0, |
| "completions/min_terminated_length": 216.0, |
| "entropy": 0.12334485203027726, |
| "epoch": 0.033582089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004527137149125338, |
| "learning_rate": 4e-05, |
| "loss": 0.0018, |
| "num_tokens": 217742.0, |
| "reward": 0.8158690929412842, |
| "reward_std": 0.14488673210144043, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.16413086652755737, |
| "rewards/length_penalty/std": 0.04868381842970848, |
| "sampling/importance_sampling_ratio/max": 1.4266942739486694, |
| "sampling/importance_sampling_ratio/mean": 0.9956211447715759, |
| "sampling/importance_sampling_ratio/min": 0.7008862495422363, |
| "sampling/sampling_logp_difference/max": 0.3554096221923828, |
| "sampling/sampling_logp_difference/mean": 0.008958252146840096, |
| "step": 9, |
| "step_time": 7.467382221948355 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002761587995337322, |
| "clip_ratio/high_mean": 0.0002761587995337322, |
| "clip_ratio/low_mean": 0.00011805738904513419, |
| "clip_ratio/low_min": 0.00011805738904513419, |
| "clip_ratio/region_mean": 0.0003942161885788664, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1656.0, |
| "completions/mean_length": 615.239990234375, |
| "completions/mean_terminated_length": 555.5416870117188, |
| "completions/min_length": 118.0, |
| "completions/min_terminated_length": 118.0, |
| "entropy": 0.1781733125448227, |
| "epoch": 0.03731343283582089, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00930685643106699, |
| "learning_rate": 4.5e-05, |
| "loss": 0.0254, |
| "num_tokens": 252474.0, |
| "reward": 0.2995898425579071, |
| "reward_std": 0.6082340478897095, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.3004101514816284, |
| "rewards/length_penalty/std": 0.23450656235218048, |
| "sampling/importance_sampling_ratio/max": 1.450623631477356, |
| "sampling/importance_sampling_ratio/mean": 0.9936089515686035, |
| "sampling/importance_sampling_ratio/min": 0.2059124857187271, |
| "sampling/sampling_logp_difference/max": 1.5803040266036987, |
| "sampling/sampling_logp_difference/mean": 0.011203057132661343, |
| "step": 10, |
| "step_time": 22.559994000708684 |
| }, |
| { |
| "clip_ratio/high_max": 0.000726152554852888, |
| "clip_ratio/high_mean": 0.000726152554852888, |
| "clip_ratio/low_mean": 0.00017200752045027913, |
| "clip_ratio/low_min": 0.00017200752045027913, |
| "clip_ratio/region_mean": 0.0008981600753031671, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 712.0, |
| "completions/max_terminated_length": 712.0, |
| "completions/mean_length": 309.0799865722656, |
| "completions/mean_terminated_length": 309.0799865722656, |
| "completions/min_length": 112.0, |
| "completions/min_terminated_length": 112.0, |
| "entropy": 0.12753532230854034, |
| "epoch": 0.041044776119402986, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005573405418545008, |
| "learning_rate": 5e-05, |
| "loss": -0.0067, |
| "num_tokens": 270938.0, |
| "reward": 0.4090820252895355, |
| "reward_std": 0.5463029742240906, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.15091796219348907, |
| "rewards/length_penalty/std": 0.08516758680343628, |
| "sampling/importance_sampling_ratio/max": 1.4622268676757812, |
| "sampling/importance_sampling_ratio/mean": 0.9957054853439331, |
| "sampling/importance_sampling_ratio/min": 0.6586275696754456, |
| "sampling/sampling_logp_difference/max": 0.41759705543518066, |
| "sampling/sampling_logp_difference/mean": 0.008718845434486866, |
| "step": 11, |
| "step_time": 8.197410132037476 |
| }, |
| { |
| "clip_ratio/high_max": 0.001248956541530788, |
| "clip_ratio/high_mean": 0.001248956541530788, |
| "clip_ratio/low_mean": 0.0002312510390765965, |
| "clip_ratio/low_min": 0.0002312510390765965, |
| "clip_ratio/region_mean": 0.001480207545682788, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1557.0, |
| "completions/mean_length": 540.8999633789062, |
| "completions/mean_terminated_length": 510.1428527832031, |
| "completions/min_length": 236.0, |
| "completions/min_terminated_length": 236.0, |
| "entropy": 0.24297742843627929, |
| "epoch": 0.04477611940298507, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011518114246428013, |
| "learning_rate": 5e-05, |
| "loss": -0.0239, |
| "num_tokens": 301663.0, |
| "reward": 0.535888671875, |
| "reward_std": 0.4949640929698944, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.26411134004592896, |
| "rewards/length_penalty/std": 0.18795593082904816, |
| "sampling/importance_sampling_ratio/max": 1.5822714567184448, |
| "sampling/importance_sampling_ratio/mean": 0.9911604523658752, |
| "sampling/importance_sampling_ratio/min": 0.6676426529884338, |
| "sampling/sampling_logp_difference/max": 0.45886147022247314, |
| "sampling/sampling_logp_difference/mean": 0.015835359692573547, |
| "step": 12, |
| "step_time": 21.649674186250195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007544187363237142, |
| "clip_ratio/high_mean": 0.0007544187363237142, |
| "clip_ratio/low_mean": 0.00023230386723298578, |
| "clip_ratio/low_min": 0.00023230386723298578, |
| "clip_ratio/region_mean": 0.0009867226035567, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1960.0, |
| "completions/mean_length": 761.8999633789062, |
| "completions/mean_terminated_length": 516.9285888671875, |
| "completions/min_length": 102.0, |
| "completions/min_terminated_length": 102.0, |
| "entropy": 0.3580703973770142, |
| "epoch": 0.048507462686567165, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011552696116268635, |
| "learning_rate": 5e-05, |
| "loss": 0.0463, |
| "num_tokens": 343388.0, |
| "reward": 0.22797851264476776, |
| "reward_std": 0.8379002213478088, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.37202149629592896, |
| "rewards/length_penalty/std": 0.36219653487205505, |
| "sampling/importance_sampling_ratio/max": 1.440017580986023, |
| "sampling/importance_sampling_ratio/mean": 0.9878596663475037, |
| "sampling/importance_sampling_ratio/min": 0.5892789363861084, |
| "sampling/sampling_logp_difference/max": 0.5288556814193726, |
| "sampling/sampling_logp_difference/mean": 0.020426517352461815, |
| "step": 13, |
| "step_time": 23.57238490600139 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006569349090568722, |
| "clip_ratio/high_mean": 0.0006569349090568722, |
| "clip_ratio/low_mean": 6.0624431353062394e-05, |
| "clip_ratio/low_min": 6.0624431353062394e-05, |
| "clip_ratio/region_mean": 0.0007175593404099345, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 497.0, |
| "completions/max_terminated_length": 497.0, |
| "completions/mean_length": 344.0799865722656, |
| "completions/mean_terminated_length": 344.0799865722656, |
| "completions/min_length": 161.0, |
| "completions/min_terminated_length": 161.0, |
| "entropy": 0.15504273474216462, |
| "epoch": 0.05223880597014925, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00864382367581129, |
| "learning_rate": 5e-05, |
| "loss": -0.0008, |
| "num_tokens": 362682.0, |
| "reward": 0.8119921684265137, |
| "reward_std": 0.14358630776405334, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.16800780594348907, |
| "rewards/length_penalty/std": 0.0466170534491539, |
| "sampling/importance_sampling_ratio/max": 1.39894700050354, |
| "sampling/importance_sampling_ratio/mean": 0.9946113228797913, |
| "sampling/importance_sampling_ratio/min": 0.6841648817062378, |
| "sampling/sampling_logp_difference/max": 0.3795562982559204, |
| "sampling/sampling_logp_difference/mean": 0.011043228209018707, |
| "step": 14, |
| "step_time": 6.20453474111855 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007670841354411095, |
| "clip_ratio/high_mean": 0.0007670841354411095, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007670841354411095, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 706.0, |
| "completions/max_terminated_length": 706.0, |
| "completions/mean_length": 319.6600036621094, |
| "completions/mean_terminated_length": 319.6600036621094, |
| "completions/min_length": 124.0, |
| "completions/min_terminated_length": 124.0, |
| "entropy": 0.160858416557312, |
| "epoch": 0.055970149253731345, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005342600867152214, |
| "learning_rate": 5e-05, |
| "loss": 0.003, |
| "num_tokens": 381195.0, |
| "reward": 0.8039159774780273, |
| "reward_std": 0.19564609229564667, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.15608398616313934, |
| "rewards/length_penalty/std": 0.05970245599746704, |
| "sampling/importance_sampling_ratio/max": 1.4084545373916626, |
| "sampling/importance_sampling_ratio/mean": 0.9941356182098389, |
| "sampling/importance_sampling_ratio/min": 0.7006469368934631, |
| "sampling/sampling_logp_difference/max": 0.3557511568069458, |
| "sampling/sampling_logp_difference/mean": 0.010944732464849949, |
| "step": 15, |
| "step_time": 7.975729608908296 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006197210110258311, |
| "clip_ratio/high_mean": 0.0006197210110258311, |
| "clip_ratio/low_mean": 2.143622696166858e-05, |
| "clip_ratio/low_min": 2.143622696166858e-05, |
| "clip_ratio/region_mean": 0.0006411572394426912, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1765.0, |
| "completions/mean_length": 642.5, |
| "completions/mean_terminated_length": 450.8409118652344, |
| "completions/min_length": 178.0, |
| "completions/min_terminated_length": 178.0, |
| "entropy": 0.16123712956905364, |
| "epoch": 0.05970149253731343, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010248241946101189, |
| "learning_rate": 5e-05, |
| "loss": 0.021, |
| "num_tokens": 416820.0, |
| "reward": 0.5462793111801147, |
| "reward_std": 0.5952073335647583, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.313720703125, |
| "rewards/length_penalty/std": 0.3165541887283325, |
| "sampling/importance_sampling_ratio/max": 1.542428970336914, |
| "sampling/importance_sampling_ratio/mean": 0.993705689907074, |
| "sampling/importance_sampling_ratio/min": 0.666167140007019, |
| "sampling/sampling_logp_difference/max": 0.43335843086242676, |
| "sampling/sampling_logp_difference/mean": 0.011193514801561832, |
| "step": 16, |
| "step_time": 22.68088135495782 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047770159144420175, |
| "clip_ratio/high_mean": 0.00047770159144420175, |
| "clip_ratio/low_mean": 4.515182226896286e-05, |
| "clip_ratio/low_min": 4.515182226896286e-05, |
| "clip_ratio/region_mean": 0.0005228534137131646, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1655.0, |
| "completions/mean_length": 748.8399658203125, |
| "completions/mean_terminated_length": 665.9148559570312, |
| "completions/min_length": 153.0, |
| "completions/min_terminated_length": 153.0, |
| "entropy": 0.1844419479370117, |
| "epoch": 0.06343283582089553, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010951795615255833, |
| "learning_rate": 5e-05, |
| "loss": 0.0253, |
| "num_tokens": 458552.0, |
| "reward": 0.29435545206069946, |
| "reward_std": 0.6421135663986206, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851815819740295, |
| "rewards/length_penalty/mean": -0.36564454436302185, |
| "rewards/length_penalty/std": 0.24128037691116333, |
| "sampling/importance_sampling_ratio/max": 1.4502041339874268, |
| "sampling/importance_sampling_ratio/mean": 0.9930887222290039, |
| "sampling/importance_sampling_ratio/min": 0.5193933844566345, |
| "sampling/sampling_logp_difference/max": 0.6550936698913574, |
| "sampling/sampling_logp_difference/mean": 0.012311195023357868, |
| "step": 17, |
| "step_time": 23.103129640687257 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004513544321525842, |
| "clip_ratio/high_mean": 0.0004513544321525842, |
| "clip_ratio/low_mean": 6.295247003436088e-05, |
| "clip_ratio/low_min": 6.295247003436088e-05, |
| "clip_ratio/region_mean": 0.0005143069021869451, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 677.0, |
| "completions/max_terminated_length": 677.0, |
| "completions/mean_length": 314.3399963378906, |
| "completions/mean_terminated_length": 314.3399963378906, |
| "completions/min_length": 160.0, |
| "completions/min_terminated_length": 160.0, |
| "entropy": 0.17831953316926957, |
| "epoch": 0.06716417910447761, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010550854727625847, |
| "learning_rate": 5e-05, |
| "loss": -0.0079, |
| "num_tokens": 477099.0, |
| "reward": 0.8265136480331421, |
| "reward_std": 0.14243507385253906, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.15348632633686066, |
| "rewards/length_penalty/std": 0.043690312653779984, |
| "sampling/importance_sampling_ratio/max": 1.4442710876464844, |
| "sampling/importance_sampling_ratio/mean": 0.9933614730834961, |
| "sampling/importance_sampling_ratio/min": 0.655369758605957, |
| "sampling/sampling_logp_difference/max": 0.42255568504333496, |
| "sampling/sampling_logp_difference/mean": 0.012269136495888233, |
| "step": 18, |
| "step_time": 8.20278396178037 |
| }, |
| { |
| "clip_ratio/high_max": 0.00040143082151189446, |
| "clip_ratio/high_mean": 0.00040143082151189446, |
| "clip_ratio/low_mean": 4.793863918166608e-05, |
| "clip_ratio/low_min": 4.793863918166608e-05, |
| "clip_ratio/region_mean": 0.0004493694577831775, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 993.0, |
| "completions/max_terminated_length": 993.0, |
| "completions/mean_length": 428.5799865722656, |
| "completions/mean_terminated_length": 428.5799865722656, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.14073365926742554, |
| "epoch": 0.0708955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007828718051314354, |
| "learning_rate": 5e-05, |
| "loss": 0.0401, |
| "num_tokens": 501738.0, |
| "reward": 0.7907323837280273, |
| "reward_std": 0.11263526231050491, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.20926757156848907, |
| "rewards/length_penalty/std": 0.11263526231050491, |
| "sampling/importance_sampling_ratio/max": 1.4394450187683105, |
| "sampling/importance_sampling_ratio/mean": 0.9950299263000488, |
| "sampling/importance_sampling_ratio/min": 0.6724317669868469, |
| "sampling/sampling_logp_difference/max": 0.3968546390533447, |
| "sampling/sampling_logp_difference/mean": 0.00956618320196867, |
| "step": 19, |
| "step_time": 11.16728618182242 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005741502449382096, |
| "clip_ratio/high_mean": 0.0005741502449382096, |
| "clip_ratio/low_mean": 0.00018747432623058557, |
| "clip_ratio/low_min": 0.00018747432623058557, |
| "clip_ratio/region_mean": 0.0007616245828103274, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1903.0, |
| "completions/mean_length": 576.6599731445312, |
| "completions/mean_terminated_length": 515.3541870117188, |
| "completions/min_length": 119.0, |
| "completions/min_terminated_length": 119.0, |
| "entropy": 0.21973758041858674, |
| "epoch": 0.07462686567164178, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013793938793241978, |
| "learning_rate": 5e-05, |
| "loss": 0.0493, |
| "num_tokens": 533481.0, |
| "reward": 0.5584277510643005, |
| "reward_std": 0.5993396639823914, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.28157225251197815, |
| "rewards/length_penalty/std": 0.26109856367111206, |
| "sampling/importance_sampling_ratio/max": 1.4702125787734985, |
| "sampling/importance_sampling_ratio/mean": 0.9921209812164307, |
| "sampling/importance_sampling_ratio/min": 0.6277052164077759, |
| "sampling/sampling_logp_difference/max": 0.4656846523284912, |
| "sampling/sampling_logp_difference/mean": 0.014733879826962948, |
| "step": 20, |
| "step_time": 22.10438237595372 |
| }, |
| { |
| "clip_ratio/high_max": 0.00031086085364222525, |
| "clip_ratio/high_mean": 0.00031086085364222525, |
| "clip_ratio/low_mean": 0.0001021033269353211, |
| "clip_ratio/low_min": 0.0001021033269353211, |
| "clip_ratio/region_mean": 0.00041296418057754634, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1112.0, |
| "completions/mean_length": 561.2000122070312, |
| "completions/mean_terminated_length": 319.16278076171875, |
| "completions/min_length": 189.0, |
| "completions/min_terminated_length": 189.0, |
| "entropy": 0.2702585756778717, |
| "epoch": 0.07835820895522388, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009658691473305225, |
| "learning_rate": 5e-05, |
| "loss": 0.0609, |
| "num_tokens": 564361.0, |
| "reward": 0.1259765625, |
| "reward_std": 0.6733714938163757, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49487167596817017, |
| "rewards/length_penalty/mean": -0.2740234434604645, |
| "rewards/length_penalty/std": 0.3115502595901489, |
| "sampling/importance_sampling_ratio/max": 1.3776147365570068, |
| "sampling/importance_sampling_ratio/mean": 0.9904045462608337, |
| "sampling/importance_sampling_ratio/min": 0.6565403342247009, |
| "sampling/sampling_logp_difference/max": 0.42077112197875977, |
| "sampling/sampling_logp_difference/mean": 0.01713278517127037, |
| "step": 21, |
| "step_time": 22.27446843800135 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004339196544606239, |
| "clip_ratio/high_mean": 0.0004339196544606239, |
| "clip_ratio/low_mean": 8.302292844746262e-05, |
| "clip_ratio/low_min": 8.302292844746262e-05, |
| "clip_ratio/region_mean": 0.0005169425858184695, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1624.0, |
| "completions/max_terminated_length": 1624.0, |
| "completions/mean_length": 403.7599792480469, |
| "completions/mean_terminated_length": 403.7599792480469, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.19005905389785765, |
| "epoch": 0.08208955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007419153116643429, |
| "learning_rate": 5e-05, |
| "loss": 0.0107, |
| "num_tokens": 587929.0, |
| "reward": 0.622851550579071, |
| "reward_std": 0.41565608978271484, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.19714844226837158, |
| "rewards/length_penalty/std": 0.18109971284866333, |
| "sampling/importance_sampling_ratio/max": 1.5093271732330322, |
| "sampling/importance_sampling_ratio/mean": 0.9929693937301636, |
| "sampling/importance_sampling_ratio/min": 0.5926903486251831, |
| "sampling/sampling_logp_difference/max": 0.5230832099914551, |
| "sampling/sampling_logp_difference/mean": 0.01249151211231947, |
| "step": 22, |
| "step_time": 17.272152825258672 |
| }, |
| { |
| "clip_ratio/high_max": 0.00048263907665386797, |
| "clip_ratio/high_mean": 0.00048263907665386797, |
| "clip_ratio/low_mean": 0.00017840272339526565, |
| "clip_ratio/low_min": 0.00017840272339526565, |
| "clip_ratio/region_mean": 0.0006610418146010488, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1995.0, |
| "completions/mean_length": 783.2799682617188, |
| "completions/mean_terminated_length": 673.3043823242188, |
| "completions/min_length": 209.0, |
| "completions/min_terminated_length": 209.0, |
| "entropy": 0.1841371864080429, |
| "epoch": 0.08582089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016199175268411636, |
| "learning_rate": 5e-05, |
| "loss": 0.0047, |
| "num_tokens": 630063.0, |
| "reward": 0.21753905713558197, |
| "reward_std": 0.6258786916732788, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.38246095180511475, |
| "rewards/length_penalty/std": 0.2692579925060272, |
| "sampling/importance_sampling_ratio/max": 1.5615938901901245, |
| "sampling/importance_sampling_ratio/mean": 0.993371307849884, |
| "sampling/importance_sampling_ratio/min": 0.6512693166732788, |
| "sampling/sampling_logp_difference/max": 0.4457070827484131, |
| "sampling/sampling_logp_difference/mean": 0.012494664639234543, |
| "step": 23, |
| "step_time": 23.557205772260204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005802198720630258, |
| "clip_ratio/high_mean": 0.0005802198720630258, |
| "clip_ratio/low_mean": 8.077544625848532e-05, |
| "clip_ratio/low_min": 8.077544625848532e-05, |
| "clip_ratio/region_mean": 0.0006609953183215111, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 902.0, |
| "completions/max_terminated_length": 902.0, |
| "completions/mean_length": 277.7599792480469, |
| "completions/mean_terminated_length": 277.7599792480469, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.21218936145305634, |
| "epoch": 0.08955223880597014, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008748773485422134, |
| "learning_rate": 5e-05, |
| "loss": -0.006, |
| "num_tokens": 646491.0, |
| "reward": 0.6043750047683716, |
| "reward_std": 0.4538353979587555, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.13562500476837158, |
| "rewards/length_penalty/std": 0.06895527988672256, |
| "sampling/importance_sampling_ratio/max": 1.4516805410385132, |
| "sampling/importance_sampling_ratio/mean": 0.992822527885437, |
| "sampling/importance_sampling_ratio/min": 0.46982115507125854, |
| "sampling/sampling_logp_difference/max": 0.7554031610488892, |
| "sampling/sampling_logp_difference/mean": 0.014432352036237717, |
| "step": 24, |
| "step_time": 9.59037149976939 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012464232742786408, |
| "clip_ratio/high_mean": 0.0012464232742786408, |
| "clip_ratio/low_mean": 7.895775488577783e-05, |
| "clip_ratio/low_min": 7.895775488577783e-05, |
| "clip_ratio/region_mean": 0.0013253810349851847, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 335.0, |
| "completions/max_terminated_length": 335.0, |
| "completions/mean_length": 219.1999969482422, |
| "completions/mean_terminated_length": 219.1999969482422, |
| "completions/min_length": 68.0, |
| "completions/min_terminated_length": 68.0, |
| "entropy": 0.17119805812835692, |
| "epoch": 0.09328358208955224, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007603586185723543, |
| "learning_rate": 5e-05, |
| "loss": 0.0056, |
| "num_tokens": 659601.0, |
| "reward": 0.8329687118530273, |
| "reward_std": 0.2540668249130249, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.10703124850988388, |
| "rewards/length_penalty/std": 0.038084786385297775, |
| "sampling/importance_sampling_ratio/max": 1.4122538566589355, |
| "sampling/importance_sampling_ratio/mean": 0.993952214717865, |
| "sampling/importance_sampling_ratio/min": 0.6891739368438721, |
| "sampling/sampling_logp_difference/max": 0.372261643409729, |
| "sampling/sampling_logp_difference/mean": 0.012025420553982258, |
| "step": 25, |
| "step_time": 4.525738164084032 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047754954139236363, |
| "clip_ratio/high_mean": 0.00047754954139236363, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00047754954139236363, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 750.0, |
| "completions/max_terminated_length": 750.0, |
| "completions/mean_length": 419.0, |
| "completions/mean_terminated_length": 419.0, |
| "completions/min_length": 133.0, |
| "completions/min_terminated_length": 133.0, |
| "entropy": 0.1435110628604889, |
| "epoch": 0.09701492537313433, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009241820313036442, |
| "learning_rate": 5e-05, |
| "loss": 0.0221, |
| "num_tokens": 683551.0, |
| "reward": 0.5354101657867432, |
| "reward_std": 0.49858328700065613, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.20458984375, |
| "rewards/length_penalty/std": 0.075383760035038, |
| "sampling/importance_sampling_ratio/max": 1.4293543100357056, |
| "sampling/importance_sampling_ratio/mean": 0.9951713681221008, |
| "sampling/importance_sampling_ratio/min": 0.672234296798706, |
| "sampling/sampling_logp_difference/max": 0.39714837074279785, |
| "sampling/sampling_logp_difference/mean": 0.009974642656743526, |
| "step": 26, |
| "step_time": 9.003170667681843 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008857163833454251, |
| "clip_ratio/high_mean": 0.0008857163833454251, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008857163833454251, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1053.0, |
| "completions/max_terminated_length": 1053.0, |
| "completions/mean_length": 459.47998046875, |
| "completions/mean_terminated_length": 459.47998046875, |
| "completions/min_length": 67.0, |
| "completions/min_terminated_length": 67.0, |
| "entropy": 0.12435056418180465, |
| "epoch": 0.10074626865671642, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012360827066004276, |
| "learning_rate": 5e-05, |
| "loss": 0.0186, |
| "num_tokens": 709325.0, |
| "reward": 0.735644519329071, |
| "reward_std": 0.26206591725349426, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.22435547411441803, |
| "rewards/length_penalty/std": 0.11961404979228973, |
| "sampling/importance_sampling_ratio/max": 1.75804603099823, |
| "sampling/importance_sampling_ratio/mean": 0.9954752922058105, |
| "sampling/importance_sampling_ratio/min": 0.6555108428001404, |
| "sampling/sampling_logp_difference/max": 0.5642030239105225, |
| "sampling/sampling_logp_difference/mean": 0.008994882926344872, |
| "step": 27, |
| "step_time": 11.897754887817428 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047460634959861636, |
| "clip_ratio/high_mean": 0.00047460634959861636, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00047460634959861636, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 802.0, |
| "completions/max_terminated_length": 802.0, |
| "completions/mean_length": 355.6199951171875, |
| "completions/mean_terminated_length": 355.6199951171875, |
| "completions/min_length": 162.0, |
| "completions/min_terminated_length": 162.0, |
| "entropy": 0.15100494623184205, |
| "epoch": 0.1044776119402985, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007981544360518456, |
| "learning_rate": 5e-05, |
| "loss": -0.0147, |
| "num_tokens": 729696.0, |
| "reward": 0.8063573837280273, |
| "reward_std": 0.16736505925655365, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1736425757408142, |
| "rewards/length_penalty/std": 0.08274995535612106, |
| "sampling/importance_sampling_ratio/max": 1.3540838956832886, |
| "sampling/importance_sampling_ratio/mean": 0.9946788549423218, |
| "sampling/importance_sampling_ratio/min": 0.6979334354400635, |
| "sampling/sampling_logp_difference/max": 0.3596315383911133, |
| "sampling/sampling_logp_difference/mean": 0.01037849672138691, |
| "step": 28, |
| "step_time": 9.409750779857859 |
| }, |
| { |
| "clip_ratio/high_max": 0.00023871174198575318, |
| "clip_ratio/high_mean": 0.00023871174198575318, |
| "clip_ratio/low_mean": 0.00039373921463266016, |
| "clip_ratio/low_min": 0.00039373921463266016, |
| "clip_ratio/region_mean": 0.0006324509595287964, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 843.0, |
| "completions/mean_length": 617.3399658203125, |
| "completions/mean_terminated_length": 303.29266357421875, |
| "completions/min_length": 140.0, |
| "completions/min_terminated_length": 140.0, |
| "entropy": 0.34604419469833375, |
| "epoch": 0.10820895522388059, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013593710958957672, |
| "learning_rate": 5e-05, |
| "loss": 0.0773, |
| "num_tokens": 763683.0, |
| "reward": 0.5185644626617432, |
| "reward_std": 0.7207738757133484, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.30143555998802185, |
| "rewards/length_penalty/std": 0.3350928723812103, |
| "sampling/importance_sampling_ratio/max": 1.4225714206695557, |
| "sampling/importance_sampling_ratio/mean": 0.9873183965682983, |
| "sampling/importance_sampling_ratio/min": 0.6167547702789307, |
| "sampling/sampling_logp_difference/max": 0.48328375816345215, |
| "sampling/sampling_logp_difference/mean": 0.021153749898076057, |
| "step": 29, |
| "step_time": 22.45293648680672 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004878065432421863, |
| "clip_ratio/high_mean": 0.0004878065432421863, |
| "clip_ratio/low_mean": 9.996298467740417e-05, |
| "clip_ratio/low_min": 9.996298467740417e-05, |
| "clip_ratio/region_mean": 0.0005877695279195905, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 670.0, |
| "completions/max_terminated_length": 670.0, |
| "completions/mean_length": 357.6600036621094, |
| "completions/mean_terminated_length": 357.6600036621094, |
| "completions/min_length": 205.0, |
| "completions/min_terminated_length": 205.0, |
| "entropy": 0.12727907598018645, |
| "epoch": 0.11194029850746269, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008195790462195873, |
| "learning_rate": 5e-05, |
| "loss": 0.0067, |
| "num_tokens": 784466.0, |
| "reward": 0.6453613042831421, |
| "reward_std": 0.4348297119140625, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.17463867366313934, |
| "rewards/length_penalty/std": 0.05574840307235718, |
| "sampling/importance_sampling_ratio/max": 1.4086308479309082, |
| "sampling/importance_sampling_ratio/mean": 0.9955052137374878, |
| "sampling/importance_sampling_ratio/min": 0.6631681323051453, |
| "sampling/sampling_logp_difference/max": 0.41072678565979004, |
| "sampling/sampling_logp_difference/mean": 0.009073731489479542, |
| "step": 30, |
| "step_time": 7.846059761708602 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008346163784153759, |
| "clip_ratio/high_mean": 0.0008346163784153759, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008346163784153759, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 901.0, |
| "completions/max_terminated_length": 901.0, |
| "completions/mean_length": 390.41998291015625, |
| "completions/mean_terminated_length": 390.41998291015625, |
| "completions/min_length": 232.0, |
| "completions/min_terminated_length": 232.0, |
| "entropy": 0.14007879197597503, |
| "epoch": 0.11567164179104478, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009134380146861076, |
| "learning_rate": 5e-05, |
| "loss": -0.0082, |
| "num_tokens": 806957.0, |
| "reward": 0.7493652105331421, |
| "reward_std": 0.24023644626140594, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979514837265, |
| "rewards/length_penalty/mean": -0.19063477218151093, |
| "rewards/length_penalty/std": 0.09261806309223175, |
| "sampling/importance_sampling_ratio/max": 1.8331904411315918, |
| "sampling/importance_sampling_ratio/mean": 0.9952817559242249, |
| "sampling/importance_sampling_ratio/min": 0.6846829056739807, |
| "sampling/sampling_logp_difference/max": 0.60605788230896, |
| "sampling/sampling_logp_difference/mean": 0.00987161509692669, |
| "step": 31, |
| "step_time": 10.142994680441916 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005887864099349827, |
| "clip_ratio/high_mean": 0.0005887864099349827, |
| "clip_ratio/low_mean": 0.00011826351401396095, |
| "clip_ratio/low_min": 0.00011826351401396095, |
| "clip_ratio/region_mean": 0.0007070499297697097, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 491.0, |
| "completions/max_terminated_length": 491.0, |
| "completions/mean_length": 348.6000061035156, |
| "completions/mean_terminated_length": 348.6000061035156, |
| "completions/min_length": 184.0, |
| "completions/min_terminated_length": 184.0, |
| "entropy": 0.16198575794696807, |
| "epoch": 0.11940298507462686, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00773199088871479, |
| "learning_rate": 5e-05, |
| "loss": 0.02, |
| "num_tokens": 826847.0, |
| "reward": 0.6297851204872131, |
| "reward_std": 0.4264633357524872, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.17021484673023224, |
| "rewards/length_penalty/std": 0.04521014913916588, |
| "sampling/importance_sampling_ratio/max": 1.4569092988967896, |
| "sampling/importance_sampling_ratio/mean": 0.9945359826087952, |
| "sampling/importance_sampling_ratio/min": 0.6699723601341248, |
| "sampling/sampling_logp_difference/max": 0.40051889419555664, |
| "sampling/sampling_logp_difference/mean": 0.011234425939619541, |
| "step": 32, |
| "step_time": 6.213606116129085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006170272710733116, |
| "clip_ratio/high_mean": 0.0006170272710733116, |
| "clip_ratio/low_mean": 0.0001180622261017561, |
| "clip_ratio/low_min": 0.0001180622261017561, |
| "clip_ratio/region_mean": 0.0007350894971750677, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1702.0, |
| "completions/max_terminated_length": 1702.0, |
| "completions/mean_length": 604.6599731445312, |
| "completions/mean_terminated_length": 604.6599731445312, |
| "completions/min_length": 153.0, |
| "completions/min_terminated_length": 153.0, |
| "entropy": 0.21591187119483948, |
| "epoch": 0.12313432835820895, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01028301753103733, |
| "learning_rate": 5e-05, |
| "loss": 0.055, |
| "num_tokens": 861810.0, |
| "reward": 0.5047558546066284, |
| "reward_std": 0.5704046487808228, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.29524412751197815, |
| "rewards/length_penalty/std": 0.19361507892608643, |
| "sampling/importance_sampling_ratio/max": 1.4207671880722046, |
| "sampling/importance_sampling_ratio/mean": 0.9924890995025635, |
| "sampling/importance_sampling_ratio/min": 0.5360985994338989, |
| "sampling/sampling_logp_difference/max": 0.6234371662139893, |
| "sampling/sampling_logp_difference/mean": 0.014163421466946602, |
| "step": 33, |
| "step_time": 19.785141061991453 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007411100086756051, |
| "clip_ratio/high_mean": 0.0007411100086756051, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007411100086756051, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 743.0, |
| "completions/max_terminated_length": 743.0, |
| "completions/mean_length": 282.1999816894531, |
| "completions/mean_terminated_length": 282.1999816894531, |
| "completions/min_length": 75.0, |
| "completions/min_terminated_length": 75.0, |
| "entropy": 0.14076103866100312, |
| "epoch": 0.12686567164179105, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0073559340089559555, |
| "learning_rate": 5e-05, |
| "loss": 0.004, |
| "num_tokens": 878860.0, |
| "reward": 0.8222070336341858, |
| "reward_std": 0.244967982172966, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.13779297471046448, |
| "rewards/length_penalty/std": 0.09619967639446259, |
| "sampling/importance_sampling_ratio/max": 1.4278913736343384, |
| "sampling/importance_sampling_ratio/mean": 0.9952441453933716, |
| "sampling/importance_sampling_ratio/min": 0.6817206144332886, |
| "sampling/sampling_logp_difference/max": 0.3831353187561035, |
| "sampling/sampling_logp_difference/mean": 0.009804938919842243, |
| "step": 34, |
| "step_time": 8.534680143930018 |
| }, |
| { |
| "clip_ratio/high_max": 0.001109963731141761, |
| "clip_ratio/high_mean": 0.001109963731141761, |
| "clip_ratio/low_mean": 0.0001634057145565748, |
| "clip_ratio/low_min": 0.0001634057145565748, |
| "clip_ratio/region_mean": 0.0012733694398775696, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1308.0, |
| "completions/mean_length": 495.94000244140625, |
| "completions/mean_terminated_length": 464.2652893066406, |
| "completions/min_length": 237.0, |
| "completions/min_terminated_length": 237.0, |
| "entropy": 0.1782270073890686, |
| "epoch": 0.13059701492537312, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009565229527652264, |
| "learning_rate": 5e-05, |
| "loss": 0.0662, |
| "num_tokens": 906687.0, |
| "reward": 0.5578417778015137, |
| "reward_std": 0.5076192021369934, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.2421582043170929, |
| "rewards/length_penalty/std": 0.14971087872982025, |
| "sampling/importance_sampling_ratio/max": 1.4500327110290527, |
| "sampling/importance_sampling_ratio/mean": 0.9938687086105347, |
| "sampling/importance_sampling_ratio/min": 0.6587642431259155, |
| "sampling/sampling_logp_difference/max": 0.4173896312713623, |
| "sampling/sampling_logp_difference/mean": 0.011907660402357578, |
| "step": 35, |
| "step_time": 21.207997231045738 |
| }, |
| { |
| "clip_ratio/high_max": 0.000516164698638022, |
| "clip_ratio/high_mean": 0.000516164698638022, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000516164698638022, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 509.0, |
| "completions/max_terminated_length": 509.0, |
| "completions/mean_length": 272.5799865722656, |
| "completions/mean_terminated_length": 272.5799865722656, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.14878032505512237, |
| "epoch": 0.13432835820895522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0072594922967255116, |
| "learning_rate": 5e-05, |
| "loss": 0.0178, |
| "num_tokens": 924236.0, |
| "reward": 0.8669042587280273, |
| "reward_std": 0.04621242731809616, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.13309569656848907, |
| "rewards/length_penalty/std": 0.04621243476867676, |
| "sampling/importance_sampling_ratio/max": 1.2858526706695557, |
| "sampling/importance_sampling_ratio/mean": 0.9944921731948853, |
| "sampling/importance_sampling_ratio/min": 0.6621037721633911, |
| "sampling/sampling_logp_difference/max": 0.41233301162719727, |
| "sampling/sampling_logp_difference/mean": 0.010732145980000496, |
| "step": 36, |
| "step_time": 6.420223196735606 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009299026743974537, |
| "clip_ratio/high_mean": 0.0009299026743974537, |
| "clip_ratio/low_mean": 2.9515937785618008e-05, |
| "clip_ratio/low_min": 2.9515937785618008e-05, |
| "clip_ratio/region_mean": 0.0009594186092726886, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1539.0, |
| "completions/mean_length": 572.1599731445312, |
| "completions/mean_terminated_length": 542.0408325195312, |
| "completions/min_length": 213.0, |
| "completions/min_terminated_length": 213.0, |
| "entropy": 0.1613151341676712, |
| "epoch": 0.13805970149253732, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020313387736678123, |
| "learning_rate": 5e-05, |
| "loss": 0.0605, |
| "num_tokens": 956624.0, |
| "reward": 0.5006250143051147, |
| "reward_std": 0.4803377687931061, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.27937498688697815, |
| "rewards/length_penalty/std": 0.18896456062793732, |
| "sampling/importance_sampling_ratio/max": 1.3775994777679443, |
| "sampling/importance_sampling_ratio/mean": 0.9941367506980896, |
| "sampling/importance_sampling_ratio/min": 0.5944056510925293, |
| "sampling/sampling_logp_difference/max": 0.520193338394165, |
| "sampling/sampling_logp_difference/mean": 0.011258355341851711, |
| "step": 37, |
| "step_time": 22.149618682917207 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007627646613400429, |
| "clip_ratio/high_mean": 0.0007627646613400429, |
| "clip_ratio/low_mean": 5.824111867696047e-05, |
| "clip_ratio/low_min": 5.824111867696047e-05, |
| "clip_ratio/region_mean": 0.0008210057800170034, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1161.0, |
| "completions/max_terminated_length": 1161.0, |
| "completions/mean_length": 346.47998046875, |
| "completions/mean_terminated_length": 346.47998046875, |
| "completions/min_length": 101.0, |
| "completions/min_terminated_length": 101.0, |
| "entropy": 0.16285331845283507, |
| "epoch": 0.1417910447761194, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006596951279789209, |
| "learning_rate": 5e-05, |
| "loss": -0.0032, |
| "num_tokens": 976898.0, |
| "reward": 0.790820300579071, |
| "reward_std": 0.21383918821811676, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.16917969286441803, |
| "rewards/length_penalty/std": 0.08555693179368973, |
| "sampling/importance_sampling_ratio/max": 1.4403197765350342, |
| "sampling/importance_sampling_ratio/mean": 0.9942667484283447, |
| "sampling/importance_sampling_ratio/min": 0.6699804663658142, |
| "sampling/sampling_logp_difference/max": 0.40050673484802246, |
| "sampling/sampling_logp_difference/mean": 0.011438596062362194, |
| "step": 38, |
| "step_time": 12.375646460102871 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006980858219321817, |
| "clip_ratio/high_mean": 0.0006980858219321817, |
| "clip_ratio/low_mean": 0.00033374534104950725, |
| "clip_ratio/low_min": 0.00033374534104950725, |
| "clip_ratio/region_mean": 0.0010318311455193908, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 553.0, |
| "completions/max_terminated_length": 553.0, |
| "completions/mean_length": 315.1999816894531, |
| "completions/mean_terminated_length": 315.1999816894531, |
| "completions/min_length": 173.0, |
| "completions/min_terminated_length": 173.0, |
| "entropy": 0.18546728193759918, |
| "epoch": 0.1455223880597015, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008189897052943707, |
| "learning_rate": 5e-05, |
| "loss": -0.0008, |
| "num_tokens": 996928.0, |
| "reward": 0.7660937309265137, |
| "reward_std": 0.275716632604599, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404752373695374, |
| "rewards/length_penalty/mean": -0.15390625596046448, |
| "rewards/length_penalty/std": 0.04370494186878204, |
| "sampling/importance_sampling_ratio/max": 1.3966577053070068, |
| "sampling/importance_sampling_ratio/mean": 0.9928783774375916, |
| "sampling/importance_sampling_ratio/min": 0.6465047001838684, |
| "sampling/sampling_logp_difference/max": 0.4361748695373535, |
| "sampling/sampling_logp_difference/mean": 0.012611854821443558, |
| "step": 39, |
| "step_time": 6.961889478377998 |
| }, |
| { |
| "clip_ratio/high_max": 0.000603052054066211, |
| "clip_ratio/high_mean": 0.000603052054066211, |
| "clip_ratio/low_mean": 3.3433633507229385e-05, |
| "clip_ratio/low_min": 3.3433633507229385e-05, |
| "clip_ratio/region_mean": 0.0006364856963045895, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1014.0, |
| "completions/max_terminated_length": 1014.0, |
| "completions/mean_length": 448.91998291015625, |
| "completions/mean_terminated_length": 448.91998291015625, |
| "completions/min_length": 114.0, |
| "completions/min_terminated_length": 114.0, |
| "entropy": 0.11928653568029404, |
| "epoch": 0.14925373134328357, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009980525821447372, |
| "learning_rate": 5e-05, |
| "loss": 0.0063, |
| "num_tokens": 1021754.0, |
| "reward": 0.5408007502555847, |
| "reward_std": 0.4783734977245331, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.21919922530651093, |
| "rewards/length_penalty/std": 0.10957939177751541, |
| "sampling/importance_sampling_ratio/max": 1.4278860092163086, |
| "sampling/importance_sampling_ratio/mean": 0.995747447013855, |
| "sampling/importance_sampling_ratio/min": 0.6380621790885925, |
| "sampling/sampling_logp_difference/max": 0.44931960105895996, |
| "sampling/sampling_logp_difference/mean": 0.008661217987537384, |
| "step": 40, |
| "step_time": 11.311517247231677 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009210823802277446, |
| "clip_ratio/high_mean": 0.0009210823802277446, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0009210823802277446, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 774.0, |
| "completions/max_terminated_length": 774.0, |
| "completions/mean_length": 379.91998291015625, |
| "completions/mean_terminated_length": 379.91998291015625, |
| "completions/min_length": 152.0, |
| "completions/min_terminated_length": 152.0, |
| "entropy": 0.19627552926540376, |
| "epoch": 0.15298507462686567, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007815813645720482, |
| "learning_rate": 5e-05, |
| "loss": 0.0047, |
| "num_tokens": 1045270.0, |
| "reward": 0.5744921565055847, |
| "reward_std": 0.4610985815525055, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.18550781905651093, |
| "rewards/length_penalty/std": 0.06156429648399353, |
| "sampling/importance_sampling_ratio/max": 1.5082297325134277, |
| "sampling/importance_sampling_ratio/mean": 0.9927440881729126, |
| "sampling/importance_sampling_ratio/min": 0.6554858684539795, |
| "sampling/sampling_logp_difference/max": 0.4223785400390625, |
| "sampling/sampling_logp_difference/mean": 0.013935217633843422, |
| "step": 41, |
| "step_time": 9.35883804736659 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003726975875906646, |
| "clip_ratio/high_mean": 0.0003726975875906646, |
| "clip_ratio/low_mean": 6.68225868139416e-05, |
| "clip_ratio/low_min": 6.68225868139416e-05, |
| "clip_ratio/region_mean": 0.0004395201802253723, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 691.0, |
| "completions/max_terminated_length": 691.0, |
| "completions/mean_length": 265.6600036621094, |
| "completions/mean_terminated_length": 265.6600036621094, |
| "completions/min_length": 48.0, |
| "completions/min_terminated_length": 48.0, |
| "entropy": 0.1420754909515381, |
| "epoch": 0.15671641791044777, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006953230127692223, |
| "learning_rate": 5e-05, |
| "loss": 0.0156, |
| "num_tokens": 1061123.0, |
| "reward": 0.8702831864356995, |
| "reward_std": 0.08517740666866302, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.12971679866313934, |
| "rewards/length_penalty/std": 0.08517740666866302, |
| "sampling/importance_sampling_ratio/max": 1.3862687349319458, |
| "sampling/importance_sampling_ratio/mean": 0.9950312972068787, |
| "sampling/importance_sampling_ratio/min": 0.6636943221092224, |
| "sampling/sampling_logp_difference/max": 0.40993356704711914, |
| "sampling/sampling_logp_difference/mean": 0.010521520860493183, |
| "step": 42, |
| "step_time": 7.802435675170273 |
| }, |
| { |
| "clip_ratio/high_max": 0.00078204206074588, |
| "clip_ratio/high_mean": 0.00078204206074588, |
| "clip_ratio/low_mean": 8.100445265881717e-05, |
| "clip_ratio/low_min": 8.100445265881717e-05, |
| "clip_ratio/region_mean": 0.0008630465308669955, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 517.0, |
| "completions/max_terminated_length": 517.0, |
| "completions/mean_length": 264.05999755859375, |
| "completions/mean_terminated_length": 264.05999755859375, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.13952410519123076, |
| "epoch": 0.16044776119402984, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008012687787413597, |
| "learning_rate": 5e-05, |
| "loss": 0.0102, |
| "num_tokens": 1077246.0, |
| "reward": 0.8510644435882568, |
| "reward_std": 0.16554366052150726, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1289355456829071, |
| "rewards/length_penalty/std": 0.04861735180020332, |
| "sampling/importance_sampling_ratio/max": 1.4597078561782837, |
| "sampling/importance_sampling_ratio/mean": 0.9948740601539612, |
| "sampling/importance_sampling_ratio/min": 0.4672364890575409, |
| "sampling/sampling_logp_difference/max": 0.7609198093414307, |
| "sampling/sampling_logp_difference/mean": 0.010898258537054062, |
| "step": 43, |
| "step_time": 6.558947047917172 |
| }, |
| { |
| "clip_ratio/high_max": 0.00034921083715744317, |
| "clip_ratio/high_mean": 0.00034921083715744317, |
| "clip_ratio/low_mean": 4.2983022285625336e-05, |
| "clip_ratio/low_min": 4.2983022285625336e-05, |
| "clip_ratio/region_mean": 0.0003921938652638346, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 787.0, |
| "completions/max_terminated_length": 787.0, |
| "completions/mean_length": 344.53997802734375, |
| "completions/mean_terminated_length": 344.53997802734375, |
| "completions/min_length": 129.0, |
| "completions/min_terminated_length": 129.0, |
| "entropy": 0.13806558549404144, |
| "epoch": 0.16417910447761194, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009876160882413387, |
| "learning_rate": 5e-05, |
| "loss": 0.0006, |
| "num_tokens": 1098703.0, |
| "reward": 0.5717675685882568, |
| "reward_std": 0.4936331808567047, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.16823242604732513, |
| "rewards/length_penalty/std": 0.10850249230861664, |
| "sampling/importance_sampling_ratio/max": 1.4370813369750977, |
| "sampling/importance_sampling_ratio/mean": 0.9950463771820068, |
| "sampling/importance_sampling_ratio/min": 0.6661191582679749, |
| "sampling/sampling_logp_difference/max": 0.40628671646118164, |
| "sampling/sampling_logp_difference/mean": 0.010221735574305058, |
| "step": 44, |
| "step_time": 9.239534565014765 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007109112164471298, |
| "clip_ratio/high_mean": 0.0007109112164471298, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007109112164471298, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 742.0, |
| "completions/max_terminated_length": 742.0, |
| "completions/mean_length": 367.17999267578125, |
| "completions/mean_terminated_length": 367.17999267578125, |
| "completions/min_length": 154.0, |
| "completions/min_terminated_length": 154.0, |
| "entropy": 0.1567250519990921, |
| "epoch": 0.16791044776119404, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008195226080715656, |
| "learning_rate": 5e-05, |
| "loss": -0.0087, |
| "num_tokens": 1120042.0, |
| "reward": 0.8007128834724426, |
| "reward_std": 0.16236166656017303, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.1792871057987213, |
| "rewards/length_penalty/std": 0.08163237571716309, |
| "sampling/importance_sampling_ratio/max": 1.4438996315002441, |
| "sampling/importance_sampling_ratio/mean": 0.9942921996116638, |
| "sampling/importance_sampling_ratio/min": 0.5561022162437439, |
| "sampling/sampling_logp_difference/max": 0.5868031978607178, |
| "sampling/sampling_logp_difference/mean": 0.011431382037699223, |
| "step": 45, |
| "step_time": 8.618486961815506 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006855419371277094, |
| "clip_ratio/high_mean": 0.0006855419371277094, |
| "clip_ratio/low_mean": 7.782101165503264e-05, |
| "clip_ratio/low_min": 7.782101165503264e-05, |
| "clip_ratio/region_mean": 0.000763362948782742, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 487.0, |
| "completions/max_terminated_length": 487.0, |
| "completions/mean_length": 249.33999633789062, |
| "completions/mean_terminated_length": 249.33999633789062, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.15599994957447053, |
| "epoch": 0.17164179104477612, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006997830234467983, |
| "learning_rate": 5e-05, |
| "loss": 0.0051, |
| "num_tokens": 1135789.0, |
| "reward": 0.8382519483566284, |
| "reward_std": 0.20857727527618408, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.12174804508686066, |
| "rewards/length_penalty/std": 0.051770467311143875, |
| "sampling/importance_sampling_ratio/max": 1.47560453414917, |
| "sampling/importance_sampling_ratio/mean": 0.9941657781600952, |
| "sampling/importance_sampling_ratio/min": 0.5952705144882202, |
| "sampling/sampling_logp_difference/max": 0.5187393426895142, |
| "sampling/sampling_logp_difference/mean": 0.012247167527675629, |
| "step": 46, |
| "step_time": 5.923294740961865 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007308489381102845, |
| "clip_ratio/high_mean": 0.0007308489381102845, |
| "clip_ratio/low_mean": 0.0002699727250728756, |
| "clip_ratio/low_min": 0.0002699727250728756, |
| "clip_ratio/region_mean": 0.001000821660272777, |
| "completions/clipped_ratio": 0.11999999731779099, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1873.0, |
| "completions/mean_length": 577.0399780273438, |
| "completions/mean_terminated_length": 376.4545593261719, |
| "completions/min_length": 144.0, |
| "completions/min_terminated_length": 144.0, |
| "entropy": 0.3190369069576263, |
| "epoch": 0.17537313432835822, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010635782033205032, |
| "learning_rate": 5e-05, |
| "loss": 0.0455, |
| "num_tokens": 1167911.0, |
| "reward": 0.3182421922683716, |
| "reward_std": 0.7228251099586487, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.28175780177116394, |
| "rewards/length_penalty/std": 0.31979885697364807, |
| "sampling/importance_sampling_ratio/max": 2.8393774032592773, |
| "sampling/importance_sampling_ratio/mean": 0.9887529611587524, |
| "sampling/importance_sampling_ratio/min": 0.6558433771133423, |
| "sampling/sampling_logp_difference/max": 1.0435848236083984, |
| "sampling/sampling_logp_difference/mean": 0.01984455995261669, |
| "step": 47, |
| "step_time": 22.49316789279692 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006271806021686643, |
| "clip_ratio/high_mean": 0.0006271806021686643, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0006271806021686643, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 758.0, |
| "completions/max_terminated_length": 758.0, |
| "completions/mean_length": 317.5799865722656, |
| "completions/mean_terminated_length": 317.5799865722656, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.16717901229858398, |
| "epoch": 0.1791044776119403, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007288388442248106, |
| "learning_rate": 5e-05, |
| "loss": -0.0015, |
| "num_tokens": 1185780.0, |
| "reward": 0.5649316310882568, |
| "reward_std": 0.41125473380088806, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.15506835281848907, |
| "rewards/length_penalty/std": 0.09044893085956573, |
| "sampling/importance_sampling_ratio/max": 2.0443358421325684, |
| "sampling/importance_sampling_ratio/mean": 0.9941331148147583, |
| "sampling/importance_sampling_ratio/min": 0.638364851474762, |
| "sampling/sampling_logp_difference/max": 0.7150728702545166, |
| "sampling/sampling_logp_difference/mean": 0.01257918868213892, |
| "step": 48, |
| "step_time": 8.94219338009134 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007135229592677206, |
| "clip_ratio/high_mean": 0.0007135229592677206, |
| "clip_ratio/low_mean": 3.603603690862656e-05, |
| "clip_ratio/low_min": 3.603603690862656e-05, |
| "clip_ratio/region_mean": 0.0007495589961763471, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1032.0, |
| "completions/max_terminated_length": 1032.0, |
| "completions/mean_length": 506.3599853515625, |
| "completions/mean_terminated_length": 506.3599853515625, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.15755688548088073, |
| "epoch": 0.1828358208955224, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009247946552932262, |
| "learning_rate": 5e-05, |
| "loss": 0.0127, |
| "num_tokens": 1213218.0, |
| "reward": 0.57275390625, |
| "reward_std": 0.45781880617141724, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.24724608659744263, |
| "rewards/length_penalty/std": 0.10344348847866058, |
| "sampling/importance_sampling_ratio/max": 1.4508453607559204, |
| "sampling/importance_sampling_ratio/mean": 0.9941213130950928, |
| "sampling/importance_sampling_ratio/min": 0.5819795727729797, |
| "sampling/sampling_logp_difference/max": 0.5413199663162231, |
| "sampling/sampling_logp_difference/mean": 0.01129181683063507, |
| "step": 49, |
| "step_time": 11.669174040900543 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008831843151710927, |
| "clip_ratio/high_mean": 0.0008831843151710927, |
| "clip_ratio/low_mean": 6.391818751581014e-05, |
| "clip_ratio/low_min": 6.391818751581014e-05, |
| "clip_ratio/region_mean": 0.0009471025085076689, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1457.0, |
| "completions/mean_length": 504.5999755859375, |
| "completions/mean_terminated_length": 473.1020202636719, |
| "completions/min_length": 121.0, |
| "completions/min_terminated_length": 121.0, |
| "entropy": 0.2619792610406876, |
| "epoch": 0.1865671641791045, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014896593056619167, |
| "learning_rate": 5e-05, |
| "loss": -0.0465, |
| "num_tokens": 1241348.0, |
| "reward": 0.47361326217651367, |
| "reward_std": 0.52699214220047, |
| "rewards/correctness/mean": 0.7200000286102295, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.24638672173023224, |
| "rewards/length_penalty/std": 0.1892664134502411, |
| "sampling/importance_sampling_ratio/max": 2.098832607269287, |
| "sampling/importance_sampling_ratio/mean": 0.9907480478286743, |
| "sampling/importance_sampling_ratio/min": 0.576706051826477, |
| "sampling/sampling_logp_difference/max": 0.7413812875747681, |
| "sampling/sampling_logp_difference/mean": 0.01703432761132717, |
| "step": 50, |
| "step_time": 21.40806119493209 |
| }, |
| { |
| "clip_ratio/high_max": 0.00019636719953268765, |
| "clip_ratio/high_mean": 0.00019636719953268765, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00019636719953268765, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 468.0, |
| "completions/max_terminated_length": 468.0, |
| "completions/mean_length": 237.37998962402344, |
| "completions/mean_terminated_length": 237.37998962402344, |
| "completions/min_length": 106.0, |
| "completions/min_terminated_length": 106.0, |
| "entropy": 0.1493792414665222, |
| "epoch": 0.19029850746268656, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007988722063601017, |
| "learning_rate": 5e-05, |
| "loss": 0.0223, |
| "num_tokens": 1255427.0, |
| "reward": 0.8840917944908142, |
| "reward_std": 0.0365288220345974, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.11590820550918579, |
| "rewards/length_penalty/std": 0.0365288220345974, |
| "sampling/importance_sampling_ratio/max": 2.037508726119995, |
| "sampling/importance_sampling_ratio/mean": 0.9946038722991943, |
| "sampling/importance_sampling_ratio/min": 0.37826406955718994, |
| "sampling/sampling_logp_difference/max": 0.9721627235412598, |
| "sampling/sampling_logp_difference/mean": 0.012720397673547268, |
| "step": 51, |
| "step_time": 5.5065665282309055 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006384580279700458, |
| "clip_ratio/high_mean": 0.0006384580279700458, |
| "clip_ratio/low_mean": 0.00015159418107941748, |
| "clip_ratio/low_min": 0.00015159418107941748, |
| "clip_ratio/region_mean": 0.0007900522090494633, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2032.0, |
| "completions/mean_length": 568.4599609375, |
| "completions/mean_terminated_length": 439.8043518066406, |
| "completions/min_length": 114.0, |
| "completions/min_terminated_length": 114.0, |
| "entropy": 0.315947163105011, |
| "epoch": 0.19402985074626866, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017853155732154846, |
| "learning_rate": 5e-05, |
| "loss": 0.0819, |
| "num_tokens": 1287180.0, |
| "reward": 0.38243162631988525, |
| "reward_std": 0.6984579563140869, |
| "rewards/correctness/mean": 0.6600000262260437, |
| "rewards/correctness/std": 0.47851812839508057, |
| "rewards/length_penalty/mean": -0.27756837010383606, |
| "rewards/length_penalty/std": 0.29989153146743774, |
| "sampling/importance_sampling_ratio/max": 1.4278913736343384, |
| "sampling/importance_sampling_ratio/mean": 0.9887783527374268, |
| "sampling/importance_sampling_ratio/min": 0.44529446959495544, |
| "sampling/sampling_logp_difference/max": 0.8090195655822754, |
| "sampling/sampling_logp_difference/mean": 0.020555591210722923, |
| "step": 52, |
| "step_time": 22.151965332916006 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007838805031497031, |
| "clip_ratio/high_mean": 0.0007838805031497031, |
| "clip_ratio/low_mean": 0.00013619677629321812, |
| "clip_ratio/low_min": 0.00013619677629321812, |
| "clip_ratio/region_mean": 0.0009200772561598569, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 479.0, |
| "completions/max_terminated_length": 479.0, |
| "completions/mean_length": 271.739990234375, |
| "completions/mean_terminated_length": 271.739990234375, |
| "completions/min_length": 131.0, |
| "completions/min_terminated_length": 131.0, |
| "entropy": 0.1576307773590088, |
| "epoch": 0.19776119402985073, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009781831875443459, |
| "learning_rate": 5e-05, |
| "loss": -0.0006, |
| "num_tokens": 1303577.0, |
| "reward": 0.7273144125938416, |
| "reward_std": 0.35877934098243713, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.13268554210662842, |
| "rewards/length_penalty/std": 0.03264938294887543, |
| "sampling/importance_sampling_ratio/max": 2.14555287361145, |
| "sampling/importance_sampling_ratio/mean": 0.9947115182876587, |
| "sampling/importance_sampling_ratio/min": 0.563351035118103, |
| "sampling/sampling_logp_difference/max": 0.763397216796875, |
| "sampling/sampling_logp_difference/mean": 0.013018809258937836, |
| "step": 53, |
| "step_time": 6.220042122993618 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009491689561400563, |
| "clip_ratio/high_mean": 0.0009491689561400563, |
| "clip_ratio/low_mean": 4.91038546897471e-05, |
| "clip_ratio/low_min": 4.91038546897471e-05, |
| "clip_ratio/region_mean": 0.0009982727991882713, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 694.0, |
| "completions/max_terminated_length": 694.0, |
| "completions/mean_length": 353.6600036621094, |
| "completions/mean_terminated_length": 353.6600036621094, |
| "completions/min_length": 142.0, |
| "completions/min_terminated_length": 142.0, |
| "entropy": 0.16784085631370543, |
| "epoch": 0.20149253731343283, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009280281141400337, |
| "learning_rate": 5e-05, |
| "loss": 0.0017, |
| "num_tokens": 1323720.0, |
| "reward": 0.26731443405151367, |
| "reward_std": 0.5046936273574829, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.17268554866313934, |
| "rewards/length_penalty/std": 0.059796757996082306, |
| "sampling/importance_sampling_ratio/max": 1.5517566204071045, |
| "sampling/importance_sampling_ratio/mean": 0.9940769076347351, |
| "sampling/importance_sampling_ratio/min": 0.5662176012992859, |
| "sampling/sampling_logp_difference/max": 0.5687768459320068, |
| "sampling/sampling_logp_difference/mean": 0.013525796122848988, |
| "step": 54, |
| "step_time": 8.069038881920278 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004341446969192475, |
| "clip_ratio/high_mean": 0.0004341446969192475, |
| "clip_ratio/low_mean": 3.74531839042902e-05, |
| "clip_ratio/low_min": 3.74531839042902e-05, |
| "clip_ratio/region_mean": 0.00047159788082353773, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 745.0, |
| "completions/mean_length": 354.47998046875, |
| "completions/mean_terminated_length": 319.9183654785156, |
| "completions/min_length": 106.0, |
| "completions/min_terminated_length": 106.0, |
| "entropy": 0.15769305527210237, |
| "epoch": 0.20522388059701493, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01787675730884075, |
| "learning_rate": 5e-05, |
| "loss": 0.0917, |
| "num_tokens": 1344304.0, |
| "reward": 0.6069140434265137, |
| "reward_std": 0.4861266314983368, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.17308594286441803, |
| "rewards/length_penalty/std": 0.1466371715068817, |
| "sampling/importance_sampling_ratio/max": 1.9533584117889404, |
| "sampling/importance_sampling_ratio/mean": 0.9941394925117493, |
| "sampling/importance_sampling_ratio/min": 0.3172576427459717, |
| "sampling/sampling_logp_difference/max": 1.1480411291122437, |
| "sampling/sampling_logp_difference/mean": 0.013562560081481934, |
| "step": 55, |
| "step_time": 20.474934197962284 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004629950679372996, |
| "clip_ratio/high_mean": 0.0004629950679372996, |
| "clip_ratio/low_mean": 4.444938385859132e-05, |
| "clip_ratio/low_min": 4.444938385859132e-05, |
| "clip_ratio/region_mean": 0.000507444451795891, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2019.0, |
| "completions/mean_length": 609.0, |
| "completions/mean_terminated_length": 334.9047546386719, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.2297052264213562, |
| "epoch": 0.208955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012723367661237717, |
| "learning_rate": 5e-05, |
| "loss": 0.0267, |
| "num_tokens": 1377194.0, |
| "reward": 0.46263670921325684, |
| "reward_std": 0.70781409740448, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.29736328125, |
| "rewards/length_penalty/std": 0.3400876820087433, |
| "sampling/importance_sampling_ratio/max": 2.268763780593872, |
| "sampling/importance_sampling_ratio/mean": 0.9911385774612427, |
| "sampling/importance_sampling_ratio/min": 0.4066545367240906, |
| "sampling/sampling_logp_difference/max": 0.8997912406921387, |
| "sampling/sampling_logp_difference/mean": 0.016253819689154625, |
| "step": 56, |
| "step_time": 22.30247456091456 |
| }, |
| { |
| "clip_ratio/high_max": 0.000757315318332985, |
| "clip_ratio/high_mean": 0.000757315318332985, |
| "clip_ratio/low_mean": 5.267316591925919e-05, |
| "clip_ratio/low_min": 5.267316591925919e-05, |
| "clip_ratio/region_mean": 0.0008099884842522442, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1166.0, |
| "completions/max_terminated_length": 1166.0, |
| "completions/mean_length": 430.9599914550781, |
| "completions/mean_terminated_length": 430.9599914550781, |
| "completions/min_length": 232.0, |
| "completions/min_terminated_length": 232.0, |
| "entropy": 0.17580842673778535, |
| "epoch": 0.2126865671641791, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011164488270878792, |
| "learning_rate": 5e-05, |
| "loss": 0.0504, |
| "num_tokens": 1402612.0, |
| "reward": 0.7895702719688416, |
| "reward_std": 0.07827731221914291, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.21042968332767487, |
| "rewards/length_penalty/std": 0.07827731221914291, |
| "sampling/importance_sampling_ratio/max": 2.252070188522339, |
| "sampling/importance_sampling_ratio/mean": 0.9938055872917175, |
| "sampling/importance_sampling_ratio/min": 0.46339696645736694, |
| "sampling/sampling_logp_difference/max": 0.81184983253479, |
| "sampling/sampling_logp_difference/mean": 0.014916970394551754, |
| "step": 57, |
| "step_time": 12.640353741589934 |
| }, |
| { |
| "clip_ratio/high_max": 0.000622326077427715, |
| "clip_ratio/high_mean": 0.000622326077427715, |
| "clip_ratio/low_mean": 8.936550584621727e-05, |
| "clip_ratio/low_min": 8.936550584621727e-05, |
| "clip_ratio/region_mean": 0.0007116915890946985, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 415.0, |
| "completions/max_terminated_length": 415.0, |
| "completions/mean_length": 223.45999145507812, |
| "completions/mean_terminated_length": 223.45999145507812, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.15358326137065886, |
| "epoch": 0.21641791044776118, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00815771147608757, |
| "learning_rate": 5e-05, |
| "loss": -0.0018, |
| "num_tokens": 1417075.0, |
| "reward": 0.5708886384963989, |
| "reward_std": 0.4770105481147766, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.10911133140325546, |
| "rewards/length_penalty/std": 0.02866869606077671, |
| "sampling/importance_sampling_ratio/max": 2.1270813941955566, |
| "sampling/importance_sampling_ratio/mean": 0.9937748312950134, |
| "sampling/importance_sampling_ratio/min": 0.23586927354335785, |
| "sampling/sampling_logp_difference/max": 1.4444775581359863, |
| "sampling/sampling_logp_difference/mean": 0.016142576932907104, |
| "step": 58, |
| "step_time": 5.718622662592679 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004469272796995938, |
| "clip_ratio/high_mean": 0.0004469272796995938, |
| "clip_ratio/low_mean": 0.000222334690624848, |
| "clip_ratio/low_min": 0.000222334690624848, |
| "clip_ratio/region_mean": 0.000669261981965974, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 431.0, |
| "completions/max_terminated_length": 431.0, |
| "completions/mean_length": 255.25999450683594, |
| "completions/mean_terminated_length": 255.25999450683594, |
| "completions/min_length": 135.0, |
| "completions/min_terminated_length": 135.0, |
| "entropy": 0.14214683771133424, |
| "epoch": 0.22014925373134328, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009184117428958416, |
| "learning_rate": 5e-05, |
| "loss": 0.0083, |
| "num_tokens": 1432748.0, |
| "reward": 0.85536128282547, |
| "reward_std": 0.15642878413200378, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.12463866919279099, |
| "rewards/length_penalty/std": 0.031104518100619316, |
| "sampling/importance_sampling_ratio/max": 2.116948366165161, |
| "sampling/importance_sampling_ratio/mean": 0.9938475489616394, |
| "sampling/importance_sampling_ratio/min": 0.36953774094581604, |
| "sampling/sampling_logp_difference/max": 0.9955024719238281, |
| "sampling/sampling_logp_difference/mean": 0.015187020413577557, |
| "step": 59, |
| "step_time": 5.274361951975152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013223550049588084, |
| "clip_ratio/high_mean": 0.0013223550049588084, |
| "clip_ratio/low_mean": 6.949270609766245e-05, |
| "clip_ratio/low_min": 6.949270609766245e-05, |
| "clip_ratio/region_mean": 0.001391847711056471, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 524.0, |
| "completions/max_terminated_length": 524.0, |
| "completions/mean_length": 310.6600036621094, |
| "completions/mean_terminated_length": 310.6600036621094, |
| "completions/min_length": 143.0, |
| "completions/min_terminated_length": 143.0, |
| "entropy": 0.1767831802368164, |
| "epoch": 0.22388059701492538, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01105369534343481, |
| "learning_rate": 5e-05, |
| "loss": -0.0052, |
| "num_tokens": 1450811.0, |
| "reward": 0.5883105397224426, |
| "reward_std": 0.4327555000782013, |
| "rewards/correctness/mean": 0.7400000095367432, |
| "rewards/correctness/std": 0.44308751821517944, |
| "rewards/length_penalty/mean": -0.15168945491313934, |
| "rewards/length_penalty/std": 0.05061161890625954, |
| "sampling/importance_sampling_ratio/max": 2.813166856765747, |
| "sampling/importance_sampling_ratio/mean": 0.9937822818756104, |
| "sampling/importance_sampling_ratio/min": 0.30162176489830017, |
| "sampling/sampling_logp_difference/max": 1.1985814571380615, |
| "sampling/sampling_logp_difference/mean": 0.01754101552069187, |
| "step": 60, |
| "step_time": 6.406686214031652 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011369908577762544, |
| "clip_ratio/high_mean": 0.0011369908577762544, |
| "clip_ratio/low_mean": 0.00026866488624364135, |
| "clip_ratio/low_min": 0.00026866488624364135, |
| "clip_ratio/region_mean": 0.0014056557440198958, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2001.0, |
| "completions/mean_length": 548.3999633789062, |
| "completions/mean_terminated_length": 418.0, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.3446581304073334, |
| "epoch": 0.22761194029850745, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006843577139079571, |
| "learning_rate": 5e-05, |
| "loss": 0.0, |
| "num_tokens": 1480661.0, |
| "reward": 0.5122265219688416, |
| "reward_std": 0.7297219634056091, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.26777344942092896, |
| "rewards/length_penalty/std": 0.32890328764915466, |
| "sampling/importance_sampling_ratio/max": 2.2495930194854736, |
| "sampling/importance_sampling_ratio/mean": 0.9877831935882568, |
| "sampling/importance_sampling_ratio/min": 0.24449598789215088, |
| "sampling/sampling_logp_difference/max": 1.4085564613342285, |
| "sampling/sampling_logp_difference/mean": 0.022903194651007652, |
| "step": 61, |
| "step_time": 22.07145461300388 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004500309267314151, |
| "clip_ratio/high_mean": 0.0004500309267314151, |
| "clip_ratio/low_mean": 0.00010034791193902493, |
| "clip_ratio/low_min": 0.00010034791193902493, |
| "clip_ratio/region_mean": 0.00055037883867044, |
| "completions/clipped_ratio": 0.14000000059604645, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1895.0, |
| "completions/mean_length": 610.3999633789062, |
| "completions/mean_terminated_length": 376.3721008300781, |
| "completions/min_length": 144.0, |
| "completions/min_terminated_length": 144.0, |
| "entropy": 0.2861504554748535, |
| "epoch": 0.23134328358208955, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017049646005034447, |
| "learning_rate": 5e-05, |
| "loss": 0.0175, |
| "num_tokens": 1513861.0, |
| "reward": 0.5219531059265137, |
| "reward_std": 0.7053924202919006, |
| "rewards/correctness/mean": 0.8199999928474426, |
| "rewards/correctness/std": 0.3880879282951355, |
| "rewards/length_penalty/mean": -0.29804688692092896, |
| "rewards/length_penalty/std": 0.3303263485431671, |
| "sampling/importance_sampling_ratio/max": 2.409658193588257, |
| "sampling/importance_sampling_ratio/mean": 0.9900499582290649, |
| "sampling/importance_sampling_ratio/min": 0.15424886345863342, |
| "sampling/sampling_logp_difference/max": 1.8691879510879517, |
| "sampling/sampling_logp_difference/mean": 0.02033442258834839, |
| "step": 62, |
| "step_time": 22.226805459707975 |
| }, |
| { |
| "clip_ratio/high_max": 0.00038249637000262735, |
| "clip_ratio/high_mean": 0.00038249637000262735, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00038249637000262735, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 902.0, |
| "completions/max_terminated_length": 902.0, |
| "completions/mean_length": 309.1999816894531, |
| "completions/mean_terminated_length": 309.1999816894531, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.16854563653469085, |
| "epoch": 0.23507462686567165, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01063678041100502, |
| "learning_rate": 5e-05, |
| "loss": 0.0004, |
| "num_tokens": 1532231.0, |
| "reward": 0.6490234136581421, |
| "reward_std": 0.4646871089935303, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.40406104922294617, |
| "rewards/length_penalty/mean": -0.15097656846046448, |
| "rewards/length_penalty/std": 0.09777788817882538, |
| "sampling/importance_sampling_ratio/max": 2.530210018157959, |
| "sampling/importance_sampling_ratio/mean": 0.9931871891021729, |
| "sampling/importance_sampling_ratio/min": 0.5236321091651917, |
| "sampling/sampling_logp_difference/max": 0.9283022880554199, |
| "sampling/sampling_logp_difference/mean": 0.01587487757205963, |
| "step": 63, |
| "step_time": 10.248050187015906 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002450101776048541, |
| "clip_ratio/high_mean": 0.0002450101776048541, |
| "clip_ratio/low_mean": 9.394082007929683e-05, |
| "clip_ratio/low_min": 9.394082007929683e-05, |
| "clip_ratio/region_mean": 0.0003389509976841509, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 636.0, |
| "completions/max_terminated_length": 636.0, |
| "completions/mean_length": 392.739990234375, |
| "completions/mean_terminated_length": 392.739990234375, |
| "completions/min_length": 204.0, |
| "completions/min_terminated_length": 204.0, |
| "entropy": 0.1435816317796707, |
| "epoch": 0.23880597014925373, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007314858492463827, |
| "learning_rate": 5e-05, |
| "loss": 0.0085, |
| "num_tokens": 1556258.0, |
| "reward": 0.7882323861122131, |
| "reward_std": 0.16649143397808075, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.19176757335662842, |
| "rewards/length_penalty/std": 0.05821884796023369, |
| "sampling/importance_sampling_ratio/max": 2.55224871635437, |
| "sampling/importance_sampling_ratio/mean": 0.9948147535324097, |
| "sampling/importance_sampling_ratio/min": 0.3057543635368347, |
| "sampling/sampling_logp_difference/max": 1.1849732398986816, |
| "sampling/sampling_logp_difference/mean": 0.014711213298141956, |
| "step": 64, |
| "step_time": 7.967395526356995 |
| }, |
| { |
| "clip_ratio/high_max": 0.00042739183409139513, |
| "clip_ratio/high_mean": 0.00042739183409139513, |
| "clip_ratio/low_mean": 0.00015360700781457125, |
| "clip_ratio/low_min": 0.00015360700781457125, |
| "clip_ratio/region_mean": 0.0005809988360852003, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 482.0, |
| "completions/max_terminated_length": 482.0, |
| "completions/mean_length": 280.1999816894531, |
| "completions/mean_terminated_length": 280.1999816894531, |
| "completions/min_length": 113.0, |
| "completions/min_terminated_length": 113.0, |
| "entropy": 0.150139519572258, |
| "epoch": 0.24253731343283583, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007787035778164864, |
| "learning_rate": 5e-05, |
| "loss": 0.0058, |
| "num_tokens": 1572738.0, |
| "reward": 0.563183605670929, |
| "reward_std": 0.49956074357032776, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.13681641221046448, |
| "rewards/length_penalty/std": 0.0541207455098629, |
| "sampling/importance_sampling_ratio/max": 2.791949510574341, |
| "sampling/importance_sampling_ratio/mean": 0.9945249557495117, |
| "sampling/importance_sampling_ratio/min": 0.16231341660022736, |
| "sampling/sampling_logp_difference/max": 1.8182260990142822, |
| "sampling/sampling_logp_difference/mean": 0.017655324190855026, |
| "step": 65, |
| "step_time": 5.998067596228793 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010637091821990908, |
| "clip_ratio/high_mean": 0.0010637091821990908, |
| "clip_ratio/low_mean": 0.00011672378168441355, |
| "clip_ratio/low_min": 0.00011672378168441355, |
| "clip_ratio/region_mean": 0.001180432946421206, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 881.0, |
| "completions/max_terminated_length": 881.0, |
| "completions/mean_length": 342.47998046875, |
| "completions/mean_terminated_length": 342.47998046875, |
| "completions/min_length": 173.0, |
| "completions/min_terminated_length": 173.0, |
| "entropy": 0.22419943511486054, |
| "epoch": 0.2462686567164179, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00808474887162447, |
| "learning_rate": 5e-05, |
| "loss": -0.0, |
| "num_tokens": 1593142.0, |
| "reward": 0.7527734041213989, |
| "reward_std": 0.27381327748298645, |
| "rewards/correctness/mean": 0.9200000166893005, |
| "rewards/correctness/std": 0.27404749393463135, |
| "rewards/length_penalty/mean": -0.16722656786441803, |
| "rewards/length_penalty/std": 0.08004416525363922, |
| "sampling/importance_sampling_ratio/max": 2.147165298461914, |
| "sampling/importance_sampling_ratio/mean": 0.9921154379844666, |
| "sampling/importance_sampling_ratio/min": 0.3174567222595215, |
| "sampling/sampling_logp_difference/max": 1.147413730621338, |
| "sampling/sampling_logp_difference/mean": 0.020227329805493355, |
| "step": 66, |
| "step_time": 10.123271165182814 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014360137982293963, |
| "clip_ratio/high_mean": 0.0014360137982293963, |
| "clip_ratio/low_mean": 0.0001180289196781814, |
| "clip_ratio/low_min": 0.0001180289196781814, |
| "clip_ratio/region_mean": 0.0015540427062660455, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1050.0, |
| "completions/max_terminated_length": 1050.0, |
| "completions/mean_length": 290.9599914550781, |
| "completions/mean_terminated_length": 290.9599914550781, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.21816076338291168, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011416135355830193, |
| "learning_rate": 5e-05, |
| "loss": 0.0122, |
| "num_tokens": 1610040.0, |
| "reward": 0.5379296541213989, |
| "reward_std": 0.506062924861908, |
| "rewards/correctness/mean": 0.6800000071525574, |
| "rewards/correctness/std": 0.4712120592594147, |
| "rewards/length_penalty/mean": -0.14207030832767487, |
| "rewards/length_penalty/std": 0.10983388125896454, |
| "sampling/importance_sampling_ratio/max": 2.0739052295684814, |
| "sampling/importance_sampling_ratio/mean": 0.9912834167480469, |
| "sampling/importance_sampling_ratio/min": 0.1838129460811615, |
| "sampling/sampling_logp_difference/max": 1.6938366889953613, |
| "sampling/sampling_logp_difference/mean": 0.020820939913392067, |
| "step": 67, |
| "step_time": 11.212650744011626 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008424129860941321, |
| "clip_ratio/high_mean": 0.0008424129860941321, |
| "clip_ratio/low_mean": 4.200357070658356e-05, |
| "clip_ratio/low_min": 4.200357070658356e-05, |
| "clip_ratio/region_mean": 0.0008844165568007156, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1909.0, |
| "completions/mean_length": 560.3800048828125, |
| "completions/mean_terminated_length": 277.0238037109375, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.3256783872842789, |
| "epoch": 0.2537313432835821, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021823778748512268, |
| "learning_rate": 5e-05, |
| "loss": 0.0302, |
| "num_tokens": 1641769.0, |
| "reward": 0.5663769245147705, |
| "reward_std": 0.7070843577384949, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.2736230492591858, |
| "rewards/length_penalty/std": 0.3544589579105377, |
| "sampling/importance_sampling_ratio/max": 2.07163405418396, |
| "sampling/importance_sampling_ratio/mean": 0.9875916838645935, |
| "sampling/importance_sampling_ratio/min": 0.19297704100608826, |
| "sampling/sampling_logp_difference/max": 1.64518404006958, |
| "sampling/sampling_logp_difference/mean": 0.02346353977918625, |
| "step": 68, |
| "step_time": 22.53464117506519 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007057235925458372, |
| "clip_ratio/high_mean": 0.0007057235925458372, |
| "clip_ratio/low_mean": 7.66577257309109e-05, |
| "clip_ratio/low_min": 7.66577257309109e-05, |
| "clip_ratio/region_mean": 0.000782381318276748, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 482.0, |
| "completions/max_terminated_length": 482.0, |
| "completions/mean_length": 233.75999450683594, |
| "completions/mean_terminated_length": 233.75999450683594, |
| "completions/min_length": 80.0, |
| "completions/min_terminated_length": 80.0, |
| "entropy": 0.18384387493133544, |
| "epoch": 0.2574626865671642, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006017310544848442, |
| "learning_rate": 5e-05, |
| "loss": -0.0044, |
| "num_tokens": 1655987.0, |
| "reward": 0.8258593678474426, |
| "reward_std": 0.23231732845306396, |
| "rewards/correctness/mean": 0.9399999976158142, |
| "rewards/correctness/std": 0.2398979365825653, |
| "rewards/length_penalty/mean": -0.11414062231779099, |
| "rewards/length_penalty/std": 0.05983913317322731, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927144050598145, |
| "sampling/importance_sampling_ratio/min": 0.16751541197299957, |
| "sampling/sampling_logp_difference/max": 1.786679983139038, |
| "sampling/sampling_logp_difference/mean": 0.020872505381703377, |
| "step": 69, |
| "step_time": 6.137518534902483 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015542366134468466, |
| "clip_ratio/high_mean": 0.0015542366134468466, |
| "clip_ratio/low_mean": 0.0001427293347660452, |
| "clip_ratio/low_min": 0.0001427293347660452, |
| "clip_ratio/region_mean": 0.001696965959854424, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 416.0, |
| "completions/mean_length": 229.77999877929688, |
| "completions/mean_terminated_length": 192.6734619140625, |
| "completions/min_length": 56.0, |
| "completions/min_terminated_length": 56.0, |
| "entropy": 0.23980919718742372, |
| "epoch": 0.26119402985074625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02051672153174877, |
| "learning_rate": 5e-05, |
| "loss": 0.1113, |
| "num_tokens": 1669906.0, |
| "reward": 0.8678027391433716, |
| "reward_std": 0.27289706468582153, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.11219726502895355, |
| "rewards/length_penalty/std": 0.13504061102867126, |
| "sampling/importance_sampling_ratio/max": 2.243411064147949, |
| "sampling/importance_sampling_ratio/mean": 0.9905449748039246, |
| "sampling/importance_sampling_ratio/min": 0.24543412029743195, |
| "sampling/sampling_logp_difference/max": 1.4047267436981201, |
| "sampling/sampling_logp_difference/mean": 0.0255883876234293, |
| "step": 70, |
| "step_time": 19.926869072951376 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010858166177058592, |
| "clip_ratio/high_mean": 0.0010858166177058592, |
| "clip_ratio/low_mean": 0.0001210609101690352, |
| "clip_ratio/low_min": 0.0001210609101690352, |
| "clip_ratio/region_mean": 0.0012068775482475757, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 716.0, |
| "completions/mean_length": 395.3399963378906, |
| "completions/mean_terminated_length": 361.61224365234375, |
| "completions/min_length": 135.0, |
| "completions/min_terminated_length": 135.0, |
| "entropy": 0.19389981627464295, |
| "epoch": 0.26492537313432835, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009714054875075817, |
| "learning_rate": 5e-05, |
| "loss": 0.0122, |
| "num_tokens": 1693373.0, |
| "reward": 0.2869628965854645, |
| "reward_std": 0.5468150973320007, |
| "rewards/correctness/mean": 0.47999998927116394, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.19303710758686066, |
| "rewards/length_penalty/std": 0.1350608468055725, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9927972555160522, |
| "sampling/importance_sampling_ratio/min": 0.13865187764167786, |
| "sampling/sampling_logp_difference/max": 1.975788950920105, |
| "sampling/sampling_logp_difference/mean": 0.01950252801179886, |
| "step": 71, |
| "step_time": 20.720340001862496 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010110618313774467, |
| "clip_ratio/high_mean": 0.0010110618313774467, |
| "clip_ratio/low_mean": 0.0002209374535595998, |
| "clip_ratio/low_min": 0.0002209374535595998, |
| "clip_ratio/region_mean": 0.0012319992762058972, |
| "completions/clipped_ratio": 0.17999999225139618, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2010.0, |
| "completions/mean_length": 934.6799926757812, |
| "completions/mean_terminated_length": 690.2926635742188, |
| "completions/min_length": 148.0, |
| "completions/min_terminated_length": 148.0, |
| "entropy": 0.3338177978992462, |
| "epoch": 0.26865671641791045, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010647455230355263, |
| "learning_rate": 5e-05, |
| "loss": 0.0436, |
| "num_tokens": 1744897.0, |
| "reward": 0.06361328065395355, |
| "reward_std": 0.7316112518310547, |
| "rewards/correctness/mean": 0.5199999809265137, |
| "rewards/correctness/std": 0.5046720504760742, |
| "rewards/length_penalty/mean": -0.4563867151737213, |
| "rewards/length_penalty/std": 0.36598309874534607, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9883248805999756, |
| "sampling/importance_sampling_ratio/min": 0.35565465688705444, |
| "sampling/sampling_logp_difference/max": 1.1421713829040527, |
| "sampling/sampling_logp_difference/mean": 0.021287694573402405, |
| "step": 72, |
| "step_time": 24.002578344894573 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015386869199573994, |
| "clip_ratio/high_mean": 0.00015386869199573994, |
| "clip_ratio/low_mean": 7.692307699471712e-05, |
| "clip_ratio/low_min": 7.692307699471712e-05, |
| "clip_ratio/region_mean": 0.00023079176899045705, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 493.0, |
| "completions/max_terminated_length": 493.0, |
| "completions/mean_length": 245.0800018310547, |
| "completions/mean_terminated_length": 245.0800018310547, |
| "completions/min_length": 62.0, |
| "completions/min_terminated_length": 62.0, |
| "entropy": 0.15735136568546296, |
| "epoch": 0.27238805970149255, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007244350388646126, |
| "learning_rate": 5e-05, |
| "loss": -0.0011, |
| "num_tokens": 1759941.0, |
| "reward": 0.84033203125, |
| "reward_std": 0.21418268978595734, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.11966796964406967, |
| "rewards/length_penalty/std": 0.04784605652093887, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9940878748893738, |
| "sampling/importance_sampling_ratio/min": 0.14858075976371765, |
| "sampling/sampling_logp_difference/max": 1.9066267013549805, |
| "sampling/sampling_logp_difference/mean": 0.020740604028105736, |
| "step": 73, |
| "step_time": 6.268040436087176 |
| }, |
| { |
| "clip_ratio/high_max": 0.00047825013170950116, |
| "clip_ratio/high_mean": 0.00047825013170950116, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00047825013170950116, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 570.0, |
| "completions/max_terminated_length": 570.0, |
| "completions/mean_length": 218.17999267578125, |
| "completions/mean_terminated_length": 218.17999267578125, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.1820651412010193, |
| "epoch": 0.27611940298507465, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00898903887718916, |
| "learning_rate": 5e-05, |
| "loss": 0.0246, |
| "num_tokens": 1773100.0, |
| "reward": 0.8934667706489563, |
| "reward_std": 0.05137346312403679, |
| "rewards/correctness/mean": 1.0, |
| "rewards/correctness/std": 0.0, |
| "rewards/length_penalty/mean": -0.10653319954872131, |
| "rewards/length_penalty/std": 0.05137345939874649, |
| "sampling/importance_sampling_ratio/max": 1.9743596315383911, |
| "sampling/importance_sampling_ratio/mean": 0.992275595664978, |
| "sampling/importance_sampling_ratio/min": 0.09230468422174454, |
| "sampling/sampling_logp_difference/max": 2.382660388946533, |
| "sampling/sampling_logp_difference/mean": 0.022644151002168655, |
| "step": 74, |
| "step_time": 6.304245180916041 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006831975246313959, |
| "clip_ratio/high_mean": 0.0006831975246313959, |
| "clip_ratio/low_mean": 0.00024684280942892655, |
| "clip_ratio/low_min": 0.00024684280942892655, |
| "clip_ratio/region_mean": 0.0009300403267843649, |
| "completions/clipped_ratio": 0.07999999821186066, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1933.0, |
| "completions/mean_length": 648.4599609375, |
| "completions/mean_terminated_length": 526.7608642578125, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.31690038442611695, |
| "epoch": 0.2798507462686567, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014707234688103199, |
| "learning_rate": 5e-05, |
| "loss": 0.0169, |
| "num_tokens": 1809483.0, |
| "reward": 0.12336913496255875, |
| "reward_std": 0.7002291083335876, |
| "rewards/correctness/mean": 0.4399999976158142, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.31663087010383606, |
| "rewards/length_penalty/std": 0.2994247078895569, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9892813563346863, |
| "sampling/importance_sampling_ratio/min": 0.2180526852607727, |
| "sampling/sampling_logp_difference/max": 1.5230185985565186, |
| "sampling/sampling_logp_difference/mean": 0.022157959640026093, |
| "step": 75, |
| "step_time": 22.700117832282558 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007681590621359646, |
| "clip_ratio/high_mean": 0.0007681590621359646, |
| "clip_ratio/low_mean": 0.0002631312469020486, |
| "clip_ratio/low_min": 0.0002631312469020486, |
| "clip_ratio/region_mean": 0.0010312903090380133, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 965.0, |
| "completions/max_terminated_length": 965.0, |
| "completions/mean_length": 287.8999938964844, |
| "completions/mean_terminated_length": 287.8999938964844, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.2121178150177002, |
| "epoch": 0.2835820895522388, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008129563182592392, |
| "learning_rate": 5e-05, |
| "loss": -0.0121, |
| "num_tokens": 1827468.0, |
| "reward": 0.4794238209724426, |
| "reward_std": 0.5341004133224487, |
| "rewards/correctness/mean": 0.6200000047683716, |
| "rewards/correctness/std": 0.4903143644332886, |
| "rewards/length_penalty/mean": -0.14057616889476776, |
| "rewards/length_penalty/std": 0.07932731509208679, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9924062490463257, |
| "sampling/importance_sampling_ratio/min": 0.05626612529158592, |
| "sampling/sampling_logp_difference/max": 2.8776626586914062, |
| "sampling/sampling_logp_difference/mean": 0.021875016391277313, |
| "step": 76, |
| "step_time": 10.302457140060142 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006995088246185332, |
| "clip_ratio/high_mean": 0.0006995088246185332, |
| "clip_ratio/low_mean": 7.473310106433927e-05, |
| "clip_ratio/low_min": 7.473310106433927e-05, |
| "clip_ratio/region_mean": 0.0007742419315036386, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2041.0, |
| "completions/mean_length": 836.3999633789062, |
| "completions/mean_terminated_length": 701.7777709960938, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.3080302506685257, |
| "epoch": 0.2873134328358209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014947418123483658, |
| "learning_rate": 5e-05, |
| "loss": 0.0184, |
| "num_tokens": 1873568.0, |
| "reward": 0.2916015684604645, |
| "reward_std": 0.6831110715866089, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4629100263118744, |
| "rewards/length_penalty/mean": -0.40839844942092896, |
| "rewards/length_penalty/std": 0.3096544146537781, |
| "sampling/importance_sampling_ratio/max": 2.289604425430298, |
| "sampling/importance_sampling_ratio/mean": 0.9889733791351318, |
| "sampling/importance_sampling_ratio/min": 0.17478086054325104, |
| "sampling/sampling_logp_difference/max": 1.7442222833633423, |
| "sampling/sampling_logp_difference/mean": 0.020468981936573982, |
| "step": 77, |
| "step_time": 23.3045789769385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011105100275017321, |
| "clip_ratio/high_mean": 0.0011105100275017321, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011105100275017321, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 423.0, |
| "completions/max_terminated_length": 423.0, |
| "completions/mean_length": 196.95999145507812, |
| "completions/mean_terminated_length": 196.95999145507812, |
| "completions/min_length": 39.0, |
| "completions/min_terminated_length": 39.0, |
| "entropy": 0.17643568515777588, |
| "epoch": 0.291044776119403, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007783598266541958, |
| "learning_rate": 5e-05, |
| "loss": 0.0026, |
| "num_tokens": 1886246.0, |
| "reward": 0.6638281345367432, |
| "reward_std": 0.44404229521751404, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.09617187827825546, |
| "rewards/length_penalty/std": 0.046398621052503586, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929860830307007, |
| "sampling/importance_sampling_ratio/min": 0.02388361468911171, |
| "sampling/sampling_logp_difference/max": 3.734562635421753, |
| "sampling/sampling_logp_difference/mean": 0.02481072209775448, |
| "step": 78, |
| "step_time": 5.516294403932989 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014786585932597518, |
| "clip_ratio/high_mean": 0.0014786585932597518, |
| "clip_ratio/low_mean": 0.00026189438067376615, |
| "clip_ratio/low_min": 0.00026189438067376615, |
| "clip_ratio/region_mean": 0.0017405529273673893, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1569.0, |
| "completions/mean_length": 473.3399963378906, |
| "completions/mean_terminated_length": 407.72918701171875, |
| "completions/min_length": 39.0, |
| "completions/min_terminated_length": 39.0, |
| "entropy": 0.22272305488586425, |
| "epoch": 0.2947761194029851, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0138373002409935, |
| "learning_rate": 5e-05, |
| "loss": 0.0374, |
| "num_tokens": 1913823.0, |
| "reward": 0.6288769245147705, |
| "reward_std": 0.4718337059020996, |
| "rewards/correctness/mean": 0.8600000143051147, |
| "rewards/correctness/std": 0.3505098223686218, |
| "rewards/length_penalty/mean": -0.23112304508686066, |
| "rewards/length_penalty/std": 0.25101178884506226, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9918286800384521, |
| "sampling/importance_sampling_ratio/min": 0.13701872527599335, |
| "sampling/sampling_logp_difference/max": 1.9876376390457153, |
| "sampling/sampling_logp_difference/mean": 0.019739534705877304, |
| "step": 79, |
| "step_time": 21.640338451368734 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018963391950819642, |
| "clip_ratio/high_mean": 0.0018963391950819642, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0018963391950819642, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 631.0, |
| "completions/max_terminated_length": 631.0, |
| "completions/mean_length": 199.6599884033203, |
| "completions/mean_terminated_length": 199.6599884033203, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.239141783118248, |
| "epoch": 0.29850746268656714, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008182297460734844, |
| "learning_rate": 5e-05, |
| "loss": 0.0016, |
| "num_tokens": 1927226.0, |
| "reward": 0.6825097799301147, |
| "reward_std": 0.4701237976551056, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.09749023616313934, |
| "rewards/length_penalty/std": 0.06250915676355362, |
| "sampling/importance_sampling_ratio/max": 2.9307050704956055, |
| "sampling/importance_sampling_ratio/mean": 0.991529643535614, |
| "sampling/importance_sampling_ratio/min": 0.2223271280527115, |
| "sampling/sampling_logp_difference/max": 1.5036054849624634, |
| "sampling/sampling_logp_difference/mean": 0.022932201623916626, |
| "step": 80, |
| "step_time": 7.0479226561728865 |
| }, |
| { |
| "clip_ratio/high_max": 0.000604342162841931, |
| "clip_ratio/high_mean": 0.000604342162841931, |
| "clip_ratio/low_mean": 6.788643368054182e-05, |
| "clip_ratio/low_min": 6.788643368054182e-05, |
| "clip_ratio/region_mean": 0.0006722285994328559, |
| "completions/clipped_ratio": 0.03999999910593033, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1783.0, |
| "completions/mean_length": 444.67999267578125, |
| "completions/mean_terminated_length": 377.875, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.2760684221982956, |
| "epoch": 0.30223880597014924, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013124141842126846, |
| "learning_rate": 5e-05, |
| "loss": 0.0099, |
| "num_tokens": 1952930.0, |
| "reward": 0.3428710997104645, |
| "reward_std": 0.6616868376731873, |
| "rewards/correctness/mean": 0.5600000023841858, |
| "rewards/correctness/std": 0.5014265179634094, |
| "rewards/length_penalty/mean": -0.2171289026737213, |
| "rewards/length_penalty/std": 0.25418272614479065, |
| "sampling/importance_sampling_ratio/max": 2.9184718132019043, |
| "sampling/importance_sampling_ratio/mean": 0.9895758628845215, |
| "sampling/importance_sampling_ratio/min": 0.04047023504972458, |
| "sampling/sampling_logp_difference/max": 3.207188606262207, |
| "sampling/sampling_logp_difference/mean": 0.024222921580076218, |
| "step": 81, |
| "step_time": 21.649420979898423 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006003081332892179, |
| "clip_ratio/high_mean": 0.0006003081332892179, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0006003081332892179, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 594.0, |
| "completions/max_terminated_length": 594.0, |
| "completions/mean_length": 184.13999938964844, |
| "completions/mean_terminated_length": 184.13999938964844, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.15603855848312378, |
| "epoch": 0.30597014925373134, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010622011497616768, |
| "learning_rate": 5e-05, |
| "loss": 0.0007, |
| "num_tokens": 1964717.0, |
| "reward": 0.6900878548622131, |
| "reward_std": 0.4795539975166321, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.08991210907697678, |
| "rewards/length_penalty/std": 0.07598428428173065, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9925838112831116, |
| "sampling/importance_sampling_ratio/min": 0.03661005198955536, |
| "sampling/sampling_logp_difference/max": 3.3074324131011963, |
| "sampling/sampling_logp_difference/mean": 0.025094155222177505, |
| "step": 82, |
| "step_time": 6.646573102101684 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009858429664745926, |
| "clip_ratio/high_mean": 0.0009858429664745926, |
| "clip_ratio/low_mean": 5.0075113540515305e-05, |
| "clip_ratio/low_min": 5.0075113540515305e-05, |
| "clip_ratio/region_mean": 0.001035918080015108, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 738.0, |
| "completions/max_terminated_length": 738.0, |
| "completions/mean_length": 331.94000244140625, |
| "completions/mean_terminated_length": 331.94000244140625, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.215926656126976, |
| "epoch": 0.30970149253731344, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009167241863906384, |
| "learning_rate": 5e-05, |
| "loss": -0.0168, |
| "num_tokens": 1983824.0, |
| "reward": 0.7979199290275574, |
| "reward_std": 0.20027850568294525, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.1620800793170929, |
| "rewards/length_penalty/std": 0.08078183978796005, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9915889501571655, |
| "sampling/importance_sampling_ratio/min": 0.205132856965065, |
| "sampling/sampling_logp_difference/max": 1.5840973854064941, |
| "sampling/sampling_logp_difference/mean": 0.01987045630812645, |
| "step": 83, |
| "step_time": 8.315541234100237 |
| }, |
| { |
| "clip_ratio/high_max": 0.002571542956866324, |
| "clip_ratio/high_mean": 0.002571542956866324, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.002571542956866324, |
| "completions/clipped_ratio": 0.19999998807907104, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 228.0, |
| "completions/mean_length": 529.5199584960938, |
| "completions/mean_terminated_length": 149.90000915527344, |
| "completions/min_length": 48.0, |
| "completions/min_terminated_length": 48.0, |
| "entropy": 0.41015579700469973, |
| "epoch": 0.31343283582089554, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006999853998422623, |
| "learning_rate": 5e-05, |
| "loss": 0.0089, |
| "num_tokens": 2013090.0, |
| "reward": 0.34144529700279236, |
| "reward_std": 0.7737928628921509, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4948716461658478, |
| "rewards/length_penalty/mean": -0.25855469703674316, |
| "rewards/length_penalty/std": 0.3750412166118622, |
| "sampling/importance_sampling_ratio/max": 2.867025852203369, |
| "sampling/importance_sampling_ratio/mean": 0.9858042001724243, |
| "sampling/importance_sampling_ratio/min": 0.041259512305259705, |
| "sampling/sampling_logp_difference/max": 3.187873601913452, |
| "sampling/sampling_logp_difference/mean": 0.02589607611298561, |
| "step": 84, |
| "step_time": 22.430606939829886 |
| }, |
| { |
| "clip_ratio/high_max": 0.001008829683996737, |
| "clip_ratio/high_mean": 0.001008829683996737, |
| "clip_ratio/low_mean": 0.0002098641009069979, |
| "clip_ratio/low_min": 0.0002098641009069979, |
| "clip_ratio/region_mean": 0.001218693784903735, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1450.0, |
| "completions/max_terminated_length": 1450.0, |
| "completions/mean_length": 248.97999572753906, |
| "completions/mean_terminated_length": 248.97999572753906, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.21633342504501343, |
| "epoch": 0.31716417910447764, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00753357307985425, |
| "learning_rate": 5e-05, |
| "loss": 0.0005, |
| "num_tokens": 2028879.0, |
| "reward": 0.638427734375, |
| "reward_std": 0.4464810788631439, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141910433769226, |
| "rewards/length_penalty/mean": -0.12157226353883743, |
| "rewards/length_penalty/std": 0.10566619038581848, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9920048117637634, |
| "sampling/importance_sampling_ratio/min": 0.21521638333797455, |
| "sampling/sampling_logp_difference/max": 1.8549904823303223, |
| "sampling/sampling_logp_difference/mean": 0.021208815276622772, |
| "step": 85, |
| "step_time": 14.633855456253514 |
| }, |
| { |
| "clip_ratio/high_max": 0.001333837426500395, |
| "clip_ratio/high_mean": 0.001333837426500395, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001333837426500395, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 495.0, |
| "completions/max_terminated_length": 495.0, |
| "completions/mean_length": 231.739990234375, |
| "completions/mean_terminated_length": 231.739990234375, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.17129210531711578, |
| "epoch": 0.3208955223880597, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008449839428067207, |
| "learning_rate": 5e-05, |
| "loss": 0.0079, |
| "num_tokens": 2042666.0, |
| "reward": 0.5268456935882568, |
| "reward_std": 0.4787832498550415, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.11315429955720901, |
| "rewards/length_penalty/std": 0.05298326164484024, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9937000274658203, |
| "sampling/importance_sampling_ratio/min": 0.08686614781618118, |
| "sampling/sampling_logp_difference/max": 2.4433867931365967, |
| "sampling/sampling_logp_difference/mean": 0.02323855087161064, |
| "step": 86, |
| "step_time": 5.850215919781476 |
| }, |
| { |
| "clip_ratio/high_max": 0.001132092683110386, |
| "clip_ratio/high_mean": 0.001132092683110386, |
| "clip_ratio/low_mean": 4.928536363877356e-05, |
| "clip_ratio/low_min": 4.928536363877356e-05, |
| "clip_ratio/region_mean": 0.0011813780525699257, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1513.0, |
| "completions/max_terminated_length": 1513.0, |
| "completions/mean_length": 339.17999267578125, |
| "completions/mean_terminated_length": 339.17999267578125, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.16397334337234498, |
| "epoch": 0.3246268656716418, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012496061623096466, |
| "learning_rate": 5e-05, |
| "loss": 0.0393, |
| "num_tokens": 2062525.0, |
| "reward": 0.794384777545929, |
| "reward_std": 0.27769869565963745, |
| "rewards/correctness/mean": 0.9599999785423279, |
| "rewards/correctness/std": 0.1979486644268036, |
| "rewards/length_penalty/mean": -0.1656152307987213, |
| "rewards/length_penalty/std": 0.1305130273103714, |
| "sampling/importance_sampling_ratio/max": 2.4161219596862793, |
| "sampling/importance_sampling_ratio/mean": 0.993759036064148, |
| "sampling/importance_sampling_ratio/min": 0.02780325338244438, |
| "sampling/sampling_logp_difference/max": 3.5826022624969482, |
| "sampling/sampling_logp_difference/mean": 0.018865756690502167, |
| "step": 87, |
| "step_time": 15.59780843090266 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006534854124765843, |
| "clip_ratio/high_mean": 0.0006534854124765843, |
| "clip_ratio/low_mean": 0.00013055354065727443, |
| "clip_ratio/low_min": 0.00013055354065727443, |
| "clip_ratio/region_mean": 0.0007840389618650079, |
| "completions/clipped_ratio": 0.05999999865889549, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1529.0, |
| "completions/mean_length": 489.2799987792969, |
| "completions/mean_terminated_length": 389.7872314453125, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.31865369975566865, |
| "epoch": 0.3283582089552239, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017925506457686424, |
| "learning_rate": 5e-05, |
| "loss": 0.0124, |
| "num_tokens": 2090599.0, |
| "reward": 0.4010937511920929, |
| "reward_std": 0.6635273098945618, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.23890624940395355, |
| "rewards/length_penalty/std": 0.26382654905319214, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9882604479789734, |
| "sampling/importance_sampling_ratio/min": 0.1960286945104599, |
| "sampling/sampling_logp_difference/max": 1.6294941902160645, |
| "sampling/sampling_logp_difference/mean": 0.023880664259195328, |
| "step": 88, |
| "step_time": 22.19856311706826 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007791827199980616, |
| "clip_ratio/high_mean": 0.0007791827199980616, |
| "clip_ratio/low_mean": 5.856515490449965e-05, |
| "clip_ratio/low_min": 5.856515490449965e-05, |
| "clip_ratio/region_mean": 0.0008377478690817953, |
| "completions/clipped_ratio": 0.019999999552965164, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1266.0, |
| "completions/mean_length": 354.3800048828125, |
| "completions/mean_terminated_length": 319.8163146972656, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.22015844881534577, |
| "epoch": 0.332089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010026220232248306, |
| "learning_rate": 5e-05, |
| "loss": 0.0039, |
| "num_tokens": 2111378.0, |
| "reward": 0.6269629001617432, |
| "reward_std": 0.4989117681980133, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.1730371117591858, |
| "rewards/length_penalty/std": 0.16782726347446442, |
| "sampling/importance_sampling_ratio/max": 2.975550651550293, |
| "sampling/importance_sampling_ratio/mean": 0.9922510981559753, |
| "sampling/importance_sampling_ratio/min": 0.0389917828142643, |
| "sampling/sampling_logp_difference/max": 3.2444043159484863, |
| "sampling/sampling_logp_difference/mean": 0.020686915144324303, |
| "step": 89, |
| "step_time": 21.06516211712733 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011358758201822639, |
| "clip_ratio/high_mean": 0.0011358758201822639, |
| "clip_ratio/low_mean": 9.847366018220783e-05, |
| "clip_ratio/low_min": 9.847366018220783e-05, |
| "clip_ratio/region_mean": 0.001234349492006004, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 670.0, |
| "completions/max_terminated_length": 670.0, |
| "completions/mean_length": 215.72000122070312, |
| "completions/mean_terminated_length": 215.72000122070312, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.18611648082733154, |
| "epoch": 0.3358208955223881, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008773848414421082, |
| "learning_rate": 5e-05, |
| "loss": 0.0001, |
| "num_tokens": 2124444.0, |
| "reward": 0.6746679544448853, |
| "reward_std": 0.41931241750717163, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519350528717, |
| "rewards/length_penalty/mean": -0.10533203184604645, |
| "rewards/length_penalty/std": 0.04260782152414322, |
| "sampling/importance_sampling_ratio/max": 2.6770498752593994, |
| "sampling/importance_sampling_ratio/mean": 0.9935617446899414, |
| "sampling/importance_sampling_ratio/min": 0.04745025932788849, |
| "sampling/sampling_logp_difference/max": 3.0480732917785645, |
| "sampling/sampling_logp_difference/mean": 0.02464917115867138, |
| "step": 90, |
| "step_time": 7.1572527338285 |
| }, |
| { |
| "clip_ratio/high_max": 0.00110072148963809, |
| "clip_ratio/high_mean": 0.00110072148963809, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00110072148963809, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 296.0, |
| "completions/max_terminated_length": 296.0, |
| "completions/mean_length": 168.239990234375, |
| "completions/mean_terminated_length": 168.239990234375, |
| "completions/min_length": 82.0, |
| "completions/min_terminated_length": 82.0, |
| "entropy": 0.1710197865962982, |
| "epoch": 0.33955223880597013, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008983846753835678, |
| "learning_rate": 5e-05, |
| "loss": 0.007, |
| "num_tokens": 2135056.0, |
| "reward": 0.8978515267372131, |
| "reward_std": 0.15232199430465698, |
| "rewards/correctness/mean": 0.9800000190734863, |
| "rewards/correctness/std": 0.1414213478565216, |
| "rewards/length_penalty/mean": -0.08214844018220901, |
| "rewards/length_penalty/std": 0.025607692077755928, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9940918684005737, |
| "sampling/importance_sampling_ratio/min": 0.10684461891651154, |
| "sampling/sampling_logp_difference/max": 2.236379623413086, |
| "sampling/sampling_logp_difference/mean": 0.02539275586605072, |
| "step": 91, |
| "step_time": 3.742940347176045 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010281951166689397, |
| "clip_ratio/high_mean": 0.0010281951166689397, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010281951166689397, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 604.0, |
| "completions/max_terminated_length": 604.0, |
| "completions/mean_length": 292.5, |
| "completions/mean_terminated_length": 292.5, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.1953798621892929, |
| "epoch": 0.34328358208955223, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007829510606825352, |
| "learning_rate": 5e-05, |
| "loss": 0.0003, |
| "num_tokens": 2153501.0, |
| "reward": 0.21717773377895355, |
| "reward_std": 0.4971000850200653, |
| "rewards/correctness/mean": 0.36000001430511475, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.142822265625, |
| "rewards/length_penalty/std": 0.05672194063663483, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9931898713111877, |
| "sampling/importance_sampling_ratio/min": 0.13717234134674072, |
| "sampling/sampling_logp_difference/max": 1.9865171909332275, |
| "sampling/sampling_logp_difference/mean": 0.02159970812499523, |
| "step": 92, |
| "step_time": 7.300173272145912 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007858758908696473, |
| "clip_ratio/high_mean": 0.0007858758908696473, |
| "clip_ratio/low_mean": 0.00013071895809844137, |
| "clip_ratio/low_min": 0.00013071895809844137, |
| "clip_ratio/region_mean": 0.0009165948489680886, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 319.0, |
| "completions/max_terminated_length": 319.0, |
| "completions/mean_length": 172.01998901367188, |
| "completions/mean_terminated_length": 172.01998901367188, |
| "completions/min_length": 100.0, |
| "completions/min_terminated_length": 100.0, |
| "entropy": 0.18890211582183838, |
| "epoch": 0.34701492537313433, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007277416530996561, |
| "learning_rate": 5e-05, |
| "loss": 0.0039, |
| "num_tokens": 2165452.0, |
| "reward": 0.7560058236122131, |
| "reward_std": 0.3761473298072815, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.08399414271116257, |
| "rewards/length_penalty/std": 0.020517654716968536, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9929280281066895, |
| "sampling/importance_sampling_ratio/min": 0.05409553647041321, |
| "sampling/sampling_logp_difference/max": 2.917003631591797, |
| "sampling/sampling_logp_difference/mean": 0.0271185040473938, |
| "step": 93, |
| "step_time": 4.145558499963954 |
| }, |
| { |
| "clip_ratio/high_max": 0.000785779458237812, |
| "clip_ratio/high_mean": 0.000785779458237812, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000785779458237812, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1327.0, |
| "completions/max_terminated_length": 1327.0, |
| "completions/mean_length": 377.5799865722656, |
| "completions/mean_terminated_length": 377.5799865722656, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.22692514657974244, |
| "epoch": 0.35074626865671643, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013048828579485416, |
| "learning_rate": 5e-05, |
| "loss": -0.0065, |
| "num_tokens": 2187441.0, |
| "reward": 0.09563476592302322, |
| "reward_std": 0.48391589522361755, |
| "rewards/correctness/mean": 0.2800000011920929, |
| "rewards/correctness/std": 0.4535573720932007, |
| "rewards/length_penalty/mean": -0.18436522781848907, |
| "rewards/length_penalty/std": 0.11944812536239624, |
| "sampling/importance_sampling_ratio/max": 2.2977609634399414, |
| "sampling/importance_sampling_ratio/mean": 0.9918870329856873, |
| "sampling/importance_sampling_ratio/min": 0.17901191115379333, |
| "sampling/sampling_logp_difference/max": 1.720302939414978, |
| "sampling/sampling_logp_difference/mean": 0.01977582275867462, |
| "step": 94, |
| "step_time": 14.208001327002421 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003195017226971686, |
| "clip_ratio/high_mean": 0.0003195017226971686, |
| "clip_ratio/low_mean": 0.0001664841634919867, |
| "clip_ratio/low_min": 0.0001664841634919867, |
| "clip_ratio/region_mean": 0.0004859858890995383, |
| "completions/clipped_ratio": 0.1599999964237213, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 895.0, |
| "completions/mean_length": 515.1799926757812, |
| "completions/mean_terminated_length": 223.21429443359375, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.14921582788228988, |
| "epoch": 0.35447761194029853, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009451565332710743, |
| "learning_rate": 5e-05, |
| "loss": 0.0568, |
| "num_tokens": 2216610.0, |
| "reward": 0.5484472513198853, |
| "reward_std": 0.7319607734680176, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4040610194206238, |
| "rewards/length_penalty/mean": -0.2515527307987213, |
| "rewards/length_penalty/std": 0.33770838379859924, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9945523738861084, |
| "sampling/importance_sampling_ratio/min": 0.15370461344718933, |
| "sampling/sampling_logp_difference/max": 1.8727226257324219, |
| "sampling/sampling_logp_difference/mean": 0.012529810890555382, |
| "step": 95, |
| "step_time": 22.457538310205564 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015879049897193909, |
| "clip_ratio/high_mean": 0.0015879049897193909, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0015879049897193909, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 693.0, |
| "completions/max_terminated_length": 693.0, |
| "completions/mean_length": 257.3800048828125, |
| "completions/mean_terminated_length": 257.3800048828125, |
| "completions/min_length": 53.0, |
| "completions/min_terminated_length": 53.0, |
| "entropy": 0.19005478024482728, |
| "epoch": 0.3582089552238806, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014516693539917469, |
| "learning_rate": 5e-05, |
| "loss": -0.0002, |
| "num_tokens": 2234099.0, |
| "reward": 0.634326159954071, |
| "reward_std": 0.4057468771934509, |
| "rewards/correctness/mean": 0.7599999904632568, |
| "rewards/correctness/std": 0.43141913414001465, |
| "rewards/length_penalty/mean": -0.1256738305091858, |
| "rewards/length_penalty/std": 0.08897071331739426, |
| "sampling/importance_sampling_ratio/max": 2.5535054206848145, |
| "sampling/importance_sampling_ratio/mean": 0.9931633472442627, |
| "sampling/importance_sampling_ratio/min": 0.1190047562122345, |
| "sampling/sampling_logp_difference/max": 2.128591775894165, |
| "sampling/sampling_logp_difference/mean": 0.019039038568735123, |
| "step": 96, |
| "step_time": 8.3912249119021 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006516657507745549, |
| "clip_ratio/high_mean": 0.0006516657507745549, |
| "clip_ratio/low_mean": 7.738037384115159e-05, |
| "clip_ratio/low_min": 7.738037384115159e-05, |
| "clip_ratio/region_mean": 0.0007290461275260895, |
| "completions/clipped_ratio": 0.09999999403953552, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2023.0, |
| "completions/mean_length": 530.4599609375, |
| "completions/mean_terminated_length": 361.8444519042969, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.18799397051334382, |
| "epoch": 0.3619402985074627, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013441347517073154, |
| "learning_rate": 5e-05, |
| "loss": 0.0231, |
| "num_tokens": 2263752.0, |
| "reward": 0.3809863328933716, |
| "reward_std": 0.6652371287345886, |
| "rewards/correctness/mean": 0.6399999856948853, |
| "rewards/correctness/std": 0.4848732352256775, |
| "rewards/length_penalty/mean": -0.25901368260383606, |
| "rewards/length_penalty/std": 0.32686105370521545, |
| "sampling/importance_sampling_ratio/max": 2.4711928367614746, |
| "sampling/importance_sampling_ratio/mean": 0.9931226968765259, |
| "sampling/importance_sampling_ratio/min": 0.15734703838825226, |
| "sampling/sampling_logp_difference/max": 1.8493014574050903, |
| "sampling/sampling_logp_difference/mean": 0.01476562675088644, |
| "step": 97, |
| "step_time": 22.094223432941362 |
| }, |
| { |
| "clip_ratio/high_max": 0.00079935536487028, |
| "clip_ratio/high_mean": 0.00079935536487028, |
| "clip_ratio/low_mean": 0.00012242774537298828, |
| "clip_ratio/low_min": 0.00012242774537298828, |
| "clip_ratio/region_mean": 0.0009217831102432683, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1552.0, |
| "completions/max_terminated_length": 1552.0, |
| "completions/mean_length": 400.03997802734375, |
| "completions/mean_terminated_length": 400.03997802734375, |
| "completions/min_length": 73.0, |
| "completions/min_terminated_length": 73.0, |
| "entropy": 0.2498596727848053, |
| "epoch": 0.3656716417910448, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012937057763338089, |
| "learning_rate": 5e-05, |
| "loss": -0.0175, |
| "num_tokens": 2287024.0, |
| "reward": 0.6446679830551147, |
| "reward_std": 0.5091517567634583, |
| "rewards/correctness/mean": 0.8399999737739563, |
| "rewards/correctness/std": 0.37032803893089294, |
| "rewards/length_penalty/mean": -0.19533203542232513, |
| "rewards/length_penalty/std": 0.19338354468345642, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9906523823738098, |
| "sampling/importance_sampling_ratio/min": 0.288949191570282, |
| "sampling/sampling_logp_difference/max": 1.241504430770874, |
| "sampling/sampling_logp_difference/mean": 0.02085532620549202, |
| "step": 98, |
| "step_time": 16.582799996016547 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012652603443711996, |
| "clip_ratio/high_mean": 0.0012652603443711996, |
| "clip_ratio/low_mean": 0.00011074523790739476, |
| "clip_ratio/low_min": 0.00011074523790739476, |
| "clip_ratio/region_mean": 0.0013760055531747638, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 741.0, |
| "completions/max_terminated_length": 741.0, |
| "completions/mean_length": 333.67999267578125, |
| "completions/mean_terminated_length": 333.67999267578125, |
| "completions/min_length": 134.0, |
| "completions/min_terminated_length": 134.0, |
| "entropy": 0.24855854511260986, |
| "epoch": 0.3694029850746269, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0096174580976367, |
| "learning_rate": 5e-05, |
| "loss": 0.007, |
| "num_tokens": 2306978.0, |
| "reward": 0.3370703160762787, |
| "reward_std": 0.5478963255882263, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5050762891769409, |
| "rewards/length_penalty/mean": -0.1629296839237213, |
| "rewards/length_penalty/std": 0.0706363394856453, |
| "sampling/importance_sampling_ratio/max": 2.512291669845581, |
| "sampling/importance_sampling_ratio/mean": 0.9905197024345398, |
| "sampling/importance_sampling_ratio/min": 0.21949462592601776, |
| "sampling/sampling_logp_difference/max": 1.5164275169372559, |
| "sampling/sampling_logp_difference/mean": 0.021251078695058823, |
| "step": 99, |
| "step_time": 8.588250870350748 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011286438559181989, |
| "clip_ratio/high_mean": 0.0011286438559181989, |
| "clip_ratio/low_mean": 0.0003009781707078218, |
| "clip_ratio/low_min": 0.0003009781707078218, |
| "clip_ratio/region_mean": 0.001429622049909085, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 389.0, |
| "completions/max_terminated_length": 389.0, |
| "completions/mean_length": 161.3000030517578, |
| "completions/mean_terminated_length": 161.3000030517578, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.16090967059135436, |
| "epoch": 0.373134328358209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006327748764306307, |
| "learning_rate": 5e-05, |
| "loss": 0.0086, |
| "num_tokens": 2317703.0, |
| "reward": 0.7012402415275574, |
| "reward_std": 0.43009650707244873, |
| "rewards/correctness/mean": 0.7799999713897705, |
| "rewards/correctness/std": 0.4184519648551941, |
| "rewards/length_penalty/mean": -0.07875976711511612, |
| "rewards/length_penalty/std": 0.03283224254846573, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9934931993484497, |
| "sampling/importance_sampling_ratio/min": 0.17191310226917267, |
| "sampling/sampling_logp_difference/max": 1.7607661485671997, |
| "sampling/sampling_logp_difference/mean": 0.020073654130101204, |
| "step": 100, |
| "step_time": 4.7784998482093215 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 200, |
| "num_input_tokens_seen": 2317703, |
| "num_train_epochs": 1, |
| "save_steps": 50, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|