| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 2.1413276231263385, |
| "eval_steps": 500, |
| "global_step": 2000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 546.9, |
| "completions/max_terminated_length": 546.9, |
| "completions/mean_length": 260.890625, |
| "completions/mean_terminated_length": 260.890625, |
| "completions/min_length": 121.9, |
| "completions/min_terminated_length": 121.9, |
| "entropy": 0.12684087702073157, |
| "epoch": 0.021413276231263382, |
| "frac_reward_zero_std": 0.75625, |
| "grad_norm": 0.1796875, |
| "learning_rate": 9.9991e-06, |
| "loss": -0.0016, |
| "num_tokens": 590852.0, |
| "reward": 0.9998437881469726, |
| "reward_std": 0.29000347256660464, |
| "rewards/reward_accuracy/mean": 0.9, |
| "rewards/reward_accuracy/std": 0.2897151708602905, |
| "rewards/reward_format/mean": 0.09984375238418579, |
| "rewards/reward_format/std": 0.0015064180828630925, |
| "sampling/importance_sampling_ratio/max": 1.7398177623748778, |
| "sampling/importance_sampling_ratio/mean": 1.007225215435028, |
| "sampling/importance_sampling_ratio/min": 0.561909693479538, |
| "sampling/sampling_logp_difference/max": 0.3685928463935852, |
| "sampling/sampling_logp_difference/mean": 0.0028177329804748297, |
| "step": 10, |
| "step_time": 7.736337069468573 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 560.8, |
| "completions/max_terminated_length": 560.8, |
| "completions/mean_length": 243.759375, |
| "completions/mean_terminated_length": 243.759375, |
| "completions/min_length": 120.1, |
| "completions/min_terminated_length": 120.1, |
| "entropy": 0.10891684428788721, |
| "epoch": 0.042826552462526764, |
| "frac_reward_zero_std": 0.76875, |
| "grad_norm": 0.298828125, |
| "learning_rate": 9.9981e-06, |
| "loss": 0.0024, |
| "num_tokens": 1159392.0, |
| "reward": 1.0053515791893006, |
| "reward_std": 0.27729729041457174, |
| "rewards/reward_accuracy/mean": 0.90546875, |
| "rewards/reward_accuracy/std": 0.27720447555184363, |
| "rewards/reward_format/mean": 0.09988281354308129, |
| "rewards/reward_format/std": 0.00132582513615489, |
| "sampling/importance_sampling_ratio/max": 1.7144915461540222, |
| "sampling/importance_sampling_ratio/mean": 0.999387800693512, |
| "sampling/importance_sampling_ratio/min": 0.5789299964904785, |
| "sampling/sampling_logp_difference/max": 0.28323344588279725, |
| "sampling/sampling_logp_difference/mean": 0.002591461222618818, |
| "step": 20, |
| "step_time": 8.23976468797773 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 620.1, |
| "completions/max_terminated_length": 620.1, |
| "completions/mean_length": 257.0546875, |
| "completions/mean_terminated_length": 257.0546875, |
| "completions/min_length": 132.3, |
| "completions/min_terminated_length": 132.3, |
| "entropy": 0.1046761798672378, |
| "epoch": 0.06423982869379015, |
| "frac_reward_zero_std": 0.6625, |
| "grad_norm": 0.30859375, |
| "learning_rate": 9.997100000000001e-06, |
| "loss": -0.0054, |
| "num_tokens": 1750702.0, |
| "reward": 0.9834375202655792, |
| "reward_std": 0.3077912166714668, |
| "rewards/reward_accuracy/mean": 0.88359375, |
| "rewards/reward_accuracy/std": 0.3075458511710167, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0017677669413387776, |
| "sampling/importance_sampling_ratio/max": 1.676391875743866, |
| "sampling/importance_sampling_ratio/mean": 1.003499412536621, |
| "sampling/importance_sampling_ratio/min": 0.510004311800003, |
| "sampling/sampling_logp_difference/max": 0.335093629360199, |
| "sampling/sampling_logp_difference/mean": 0.002611653762869537, |
| "step": 30, |
| "step_time": 8.064201896078885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 669.6, |
| "completions/max_terminated_length": 669.6, |
| "completions/mean_length": 251.60859375, |
| "completions/mean_terminated_length": 251.60859375, |
| "completions/min_length": 117.4, |
| "completions/min_terminated_length": 117.4, |
| "entropy": 0.10029142308048904, |
| "epoch": 0.08565310492505353, |
| "frac_reward_zero_std": 0.725, |
| "grad_norm": 0.255859375, |
| "learning_rate": 9.9961e-06, |
| "loss": -0.0078, |
| "num_tokens": 2331841.0, |
| "reward": 1.004687523841858, |
| "reward_std": 0.28303155973553656, |
| "rewards/reward_accuracy/mean": 0.9046875, |
| "rewards/reward_accuracy/std": 0.28303155973553656, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7131255507469176, |
| "sampling/importance_sampling_ratio/mean": 0.9939331233501434, |
| "sampling/importance_sampling_ratio/min": 0.5193794310092926, |
| "sampling/sampling_logp_difference/max": 0.2736384391784668, |
| "sampling/sampling_logp_difference/mean": 0.0026209764182567596, |
| "step": 40, |
| "step_time": 9.051525293383747 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 571.4, |
| "completions/max_terminated_length": 571.4, |
| "completions/mean_length": 229.35078125, |
| "completions/mean_terminated_length": 229.35078125, |
| "completions/min_length": 117.9, |
| "completions/min_terminated_length": 117.9, |
| "entropy": 0.09306560978293418, |
| "epoch": 0.10706638115631692, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.1328125, |
| "learning_rate": 9.9951e-06, |
| "loss": -0.0, |
| "num_tokens": 2880274.0, |
| "reward": 1.020312523841858, |
| "reward_std": 0.25418364331126214, |
| "rewards/reward_accuracy/mean": 0.9203125, |
| "rewards/reward_accuracy/std": 0.25418364331126214, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6562096118927, |
| "sampling/importance_sampling_ratio/mean": 0.9955496251583099, |
| "sampling/importance_sampling_ratio/min": 0.591128158569336, |
| "sampling/sampling_logp_difference/max": 0.3064452469348907, |
| "sampling/sampling_logp_difference/mean": 0.002502969070337713, |
| "step": 50, |
| "step_time": 7.7682885531336066 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 479.6, |
| "completions/max_terminated_length": 479.6, |
| "completions/mean_length": 238.41484375, |
| "completions/mean_terminated_length": 238.41484375, |
| "completions/min_length": 119.3, |
| "completions/min_terminated_length": 119.3, |
| "entropy": 0.09468000261113048, |
| "epoch": 0.1284796573875803, |
| "frac_reward_zero_std": 0.75625, |
| "grad_norm": 0.3203125, |
| "learning_rate": 9.994100000000001e-06, |
| "loss": 0.0038, |
| "num_tokens": 3443173.0, |
| "reward": 1.014062523841858, |
| "reward_std": 0.25716659501194955, |
| "rewards/reward_accuracy/mean": 0.9140625, |
| "rewards/reward_accuracy/std": 0.25716659501194955, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7249961376190186, |
| "sampling/importance_sampling_ratio/mean": 1.0005383729934691, |
| "sampling/importance_sampling_ratio/min": 0.558535149693489, |
| "sampling/sampling_logp_difference/max": 0.2923584520816803, |
| "sampling/sampling_logp_difference/mean": 0.0026324418606236575, |
| "step": 60, |
| "step_time": 7.220544941257685 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 541.3, |
| "completions/max_terminated_length": 541.3, |
| "completions/mean_length": 248.10625, |
| "completions/mean_terminated_length": 248.10625, |
| "completions/min_length": 111.4, |
| "completions/min_terminated_length": 111.4, |
| "entropy": 0.08918840945698321, |
| "epoch": 0.14989293361884368, |
| "frac_reward_zero_std": 0.78125, |
| "grad_norm": 0.42578125, |
| "learning_rate": 9.993100000000001e-06, |
| "loss": -0.0013, |
| "num_tokens": 4021477.0, |
| "reward": 1.021875023841858, |
| "reward_std": 0.24348516762256622, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.243485164642334, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7035181283950807, |
| "sampling/importance_sampling_ratio/mean": 0.9972749769687652, |
| "sampling/importance_sampling_ratio/min": 0.48475154396146536, |
| "sampling/sampling_logp_difference/max": 0.6903421759605408, |
| "sampling/sampling_logp_difference/mean": 0.002532821847125888, |
| "step": 70, |
| "step_time": 7.656073545431719 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 524.4, |
| "completions/max_terminated_length": 524.4, |
| "completions/mean_length": 245.90078125, |
| "completions/mean_terminated_length": 245.90078125, |
| "completions/min_length": 123.9, |
| "completions/min_terminated_length": 123.9, |
| "entropy": 0.09812549804337323, |
| "epoch": 0.17130620985010706, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.1396484375, |
| "learning_rate": 9.9921e-06, |
| "loss": -0.0039, |
| "num_tokens": 4593734.0, |
| "reward": 1.029687523841858, |
| "reward_std": 0.24289729669690133, |
| "rewards/reward_accuracy/mean": 0.9296875, |
| "rewards/reward_accuracy/std": 0.24289729669690133, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7347792744636537, |
| "sampling/importance_sampling_ratio/mean": 0.9934443175792694, |
| "sampling/importance_sampling_ratio/min": 0.5508507907390594, |
| "sampling/sampling_logp_difference/max": 0.3118584156036377, |
| "sampling/sampling_logp_difference/mean": 0.0027633092366158964, |
| "step": 80, |
| "step_time": 7.578580120345578 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 578.2, |
| "completions/max_terminated_length": 578.2, |
| "completions/mean_length": 256.69765625, |
| "completions/mean_terminated_length": 256.69765625, |
| "completions/min_length": 123.1, |
| "completions/min_terminated_length": 123.1, |
| "entropy": 0.09923308668658137, |
| "epoch": 0.19271948608137046, |
| "frac_reward_zero_std": 0.775, |
| "grad_norm": 0.2099609375, |
| "learning_rate": 9.991100000000002e-06, |
| "loss": -0.004, |
| "num_tokens": 5180595.0, |
| "reward": 0.9945312738418579, |
| "reward_std": 0.2973235800862312, |
| "rewards/reward_accuracy/mean": 0.89453125, |
| "rewards/reward_accuracy/std": 0.297323577105999, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8877214312553405, |
| "sampling/importance_sampling_ratio/mean": 0.9834304213523865, |
| "sampling/importance_sampling_ratio/min": 0.5366867035627365, |
| "sampling/sampling_logp_difference/max": 0.31917338371276854, |
| "sampling/sampling_logp_difference/mean": 0.0027826699428260327, |
| "step": 90, |
| "step_time": 7.933970131957904 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 639.0, |
| "completions/max_terminated_length": 639.0, |
| "completions/mean_length": 269.81171875, |
| "completions/mean_terminated_length": 269.81171875, |
| "completions/min_length": 134.8, |
| "completions/min_terminated_length": 134.8, |
| "entropy": 0.10335401892662048, |
| "epoch": 0.21413276231263384, |
| "frac_reward_zero_std": 0.79375, |
| "grad_norm": 0.2314453125, |
| "learning_rate": 9.990100000000001e-06, |
| "loss": 0.0063, |
| "num_tokens": 5785818.0, |
| "reward": 1.008593773841858, |
| "reward_std": 0.27817061841487883, |
| "rewards/reward_accuracy/mean": 0.90859375, |
| "rewards/reward_accuracy/std": 0.27817061841487883, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.822286307811737, |
| "sampling/importance_sampling_ratio/mean": 1.0062408030033112, |
| "sampling/importance_sampling_ratio/min": 0.5896649330854415, |
| "sampling/sampling_logp_difference/max": 0.3775724768638611, |
| "sampling/sampling_logp_difference/mean": 0.002754453872330487, |
| "step": 100, |
| "step_time": 8.422840794082731 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 597.6, |
| "completions/max_terminated_length": 597.6, |
| "completions/mean_length": 264.75703125, |
| "completions/mean_terminated_length": 264.75703125, |
| "completions/min_length": 140.6, |
| "completions/min_terminated_length": 140.6, |
| "entropy": 0.11051499620079994, |
| "epoch": 0.23554603854389722, |
| "frac_reward_zero_std": 0.80625, |
| "grad_norm": 0.298828125, |
| "learning_rate": 9.9891e-06, |
| "loss": -0.0036, |
| "num_tokens": 6383835.0, |
| "reward": 1.028867208957672, |
| "reward_std": 0.22532478724606336, |
| "rewards/reward_accuracy/mean": 0.92890625, |
| "rewards/reward_accuracy/std": 0.22488284558057786, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.8296730160713195, |
| "sampling/importance_sampling_ratio/mean": 0.9965243399143219, |
| "sampling/importance_sampling_ratio/min": 0.5352214843034744, |
| "sampling/sampling_logp_difference/max": 0.2701212286949158, |
| "sampling/sampling_logp_difference/mean": 0.002912667999044061, |
| "step": 110, |
| "step_time": 8.348192108981312 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 618.4, |
| "completions/max_terminated_length": 618.4, |
| "completions/mean_length": 280.66328125, |
| "completions/mean_terminated_length": 280.66328125, |
| "completions/min_length": 138.2, |
| "completions/min_terminated_length": 138.2, |
| "entropy": 0.11365606668405234, |
| "epoch": 0.2569593147751606, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.228515625, |
| "learning_rate": 9.9881e-06, |
| "loss": 0.0009, |
| "num_tokens": 7003884.0, |
| "reward": 1.0117187738418578, |
| "reward_std": 0.26884681433439256, |
| "rewards/reward_accuracy/mean": 0.91171875, |
| "rewards/reward_accuracy/std": 0.26884681135416033, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7911346673965454, |
| "sampling/importance_sampling_ratio/mean": 1.0083366215229035, |
| "sampling/importance_sampling_ratio/min": 0.5375795006752014, |
| "sampling/sampling_logp_difference/max": 0.3476578712463379, |
| "sampling/sampling_logp_difference/mean": 0.002910136035643518, |
| "step": 120, |
| "step_time": 8.220505654579028 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 662.6, |
| "completions/max_terminated_length": 662.6, |
| "completions/mean_length": 270.05859375, |
| "completions/mean_terminated_length": 270.05859375, |
| "completions/min_length": 131.5, |
| "completions/min_terminated_length": 131.5, |
| "entropy": 0.11095834248699248, |
| "epoch": 0.278372591006424, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.2041015625, |
| "learning_rate": 9.987100000000001e-06, |
| "loss": 0.0021, |
| "num_tokens": 7606567.0, |
| "reward": 1.009335958957672, |
| "reward_std": 0.27545640170574187, |
| "rewards/reward_accuracy/mean": 0.909375, |
| "rewards/reward_accuracy/std": 0.27547004222869875, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.8426976203918457, |
| "sampling/importance_sampling_ratio/mean": 1.008215057849884, |
| "sampling/importance_sampling_ratio/min": 0.4933893457055092, |
| "sampling/sampling_logp_difference/max": 0.36618590354919434, |
| "sampling/sampling_logp_difference/mean": 0.0028675782727077604, |
| "step": 130, |
| "step_time": 8.578948781173676 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 759.8, |
| "completions/max_terminated_length": 759.8, |
| "completions/mean_length": 275.740625, |
| "completions/mean_terminated_length": 275.740625, |
| "completions/min_length": 132.9, |
| "completions/min_terminated_length": 132.9, |
| "entropy": 0.10890614441595972, |
| "epoch": 0.29978586723768735, |
| "frac_reward_zero_std": 0.7, |
| "grad_norm": 0.21875, |
| "learning_rate": 9.9861e-06, |
| "loss": -0.0015, |
| "num_tokens": 8219635.0, |
| "reward": 0.977226585149765, |
| "reward_std": 0.31434632688760755, |
| "rewards/reward_accuracy/mean": 0.87734375, |
| "rewards/reward_accuracy/std": 0.31417865604162215, |
| "rewards/reward_format/mean": 0.09988281428813935, |
| "rewards/reward_format/std": 0.0013258252292871475, |
| "sampling/importance_sampling_ratio/max": 1.6868999481201172, |
| "sampling/importance_sampling_ratio/mean": 0.9944325923919678, |
| "sampling/importance_sampling_ratio/min": 0.5429329872131348, |
| "sampling/sampling_logp_difference/max": 0.26084762811660767, |
| "sampling/sampling_logp_difference/mean": 0.0029025043593719603, |
| "step": 140, |
| "step_time": 9.44805132704787 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 675.5, |
| "completions/max_terminated_length": 675.5, |
| "completions/mean_length": 271.96875, |
| "completions/mean_terminated_length": 271.96875, |
| "completions/min_length": 129.5, |
| "completions/min_terminated_length": 129.5, |
| "entropy": 0.10716174631379545, |
| "epoch": 0.32119914346895073, |
| "frac_reward_zero_std": 0.75625, |
| "grad_norm": 0.396484375, |
| "learning_rate": 9.9851e-06, |
| "loss": 0.0021, |
| "num_tokens": 8826539.0, |
| "reward": 0.997656273841858, |
| "reward_std": 0.29206475168466567, |
| "rewards/reward_accuracy/mean": 0.89765625, |
| "rewards/reward_accuracy/std": 0.29206475168466567, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.932285439968109, |
| "sampling/importance_sampling_ratio/mean": 1.005907541513443, |
| "sampling/importance_sampling_ratio/min": 0.4102198973298073, |
| "sampling/sampling_logp_difference/max": 0.3344579219818115, |
| "sampling/sampling_logp_difference/mean": 0.0028855334036052226, |
| "step": 150, |
| "step_time": 8.742801927216352 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 598.3, |
| "completions/max_terminated_length": 598.3, |
| "completions/mean_length": 265.15625, |
| "completions/mean_terminated_length": 265.15625, |
| "completions/min_length": 130.5, |
| "completions/min_terminated_length": 130.5, |
| "entropy": 0.11050200234167278, |
| "epoch": 0.3426124197002141, |
| "frac_reward_zero_std": 0.74375, |
| "grad_norm": 0.21875, |
| "learning_rate": 9.984100000000002e-06, |
| "loss": -0.0046, |
| "num_tokens": 9424075.0, |
| "reward": 0.9991406500339508, |
| "reward_std": 0.276620976626873, |
| "rewards/reward_accuracy/mean": 0.89921875, |
| "rewards/reward_accuracy/std": 0.27662705108523367, |
| "rewards/reward_format/mean": 0.09992187619209289, |
| "rewards/reward_format/std": 0.00088388342410326, |
| "sampling/importance_sampling_ratio/max": 1.828334355354309, |
| "sampling/importance_sampling_ratio/mean": 0.9949215412139892, |
| "sampling/importance_sampling_ratio/min": 0.521588608622551, |
| "sampling/sampling_logp_difference/max": 0.310284423828125, |
| "sampling/sampling_logp_difference/mean": 0.0029434302588924764, |
| "step": 160, |
| "step_time": 8.040386501932517 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 886.7, |
| "completions/max_terminated_length": 701.2, |
| "completions/mean_length": 276.26328125, |
| "completions/mean_terminated_length": 274.0525161743164, |
| "completions/min_length": 132.8, |
| "completions/min_terminated_length": 132.8, |
| "entropy": 0.11025537732057274, |
| "epoch": 0.3640256959314775, |
| "frac_reward_zero_std": 0.75625, |
| "grad_norm": 0.259765625, |
| "learning_rate": 9.983100000000001e-06, |
| "loss": -0.0015, |
| "num_tokens": 10039236.0, |
| "reward": 0.9741406559944152, |
| "reward_std": 0.3239198952913284, |
| "rewards/reward_accuracy/mean": 0.87421875, |
| "rewards/reward_accuracy/std": 0.3235792338848114, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.864332103729248, |
| "sampling/importance_sampling_ratio/mean": 1.0021799981594086, |
| "sampling/importance_sampling_ratio/min": 0.42256206199526786, |
| "sampling/sampling_logp_difference/max": 0.3006327271461487, |
| "sampling/sampling_logp_difference/mean": 0.0029476020019501446, |
| "step": 170, |
| "step_time": 10.88967755115591 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 694.5, |
| "completions/max_terminated_length": 694.5, |
| "completions/mean_length": 278.79375, |
| "completions/mean_terminated_length": 278.79375, |
| "completions/min_length": 134.8, |
| "completions/min_terminated_length": 134.8, |
| "entropy": 0.11467711511068046, |
| "epoch": 0.3854389721627409, |
| "frac_reward_zero_std": 0.69375, |
| "grad_norm": 0.236328125, |
| "learning_rate": 9.9821e-06, |
| "loss": -0.0024, |
| "num_tokens": 10657228.0, |
| "reward": 0.9788281440734863, |
| "reward_std": 0.32416844069957734, |
| "rewards/reward_accuracy/mean": 0.87890625, |
| "rewards/reward_accuracy/std": 0.3241883754730225, |
| "rewards/reward_format/mean": 0.09992187619209289, |
| "rewards/reward_format/std": 0.00088388342410326, |
| "sampling/importance_sampling_ratio/max": 1.8137566208839417, |
| "sampling/importance_sampling_ratio/mean": 1.0055959284305573, |
| "sampling/importance_sampling_ratio/min": 0.4928945809602737, |
| "sampling/sampling_logp_difference/max": 0.36681064367294314, |
| "sampling/sampling_logp_difference/mean": 0.003075344511307776, |
| "step": 180, |
| "step_time": 8.963536798162385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 734.1, |
| "completions/max_terminated_length": 734.1, |
| "completions/mean_length": 269.8953125, |
| "completions/mean_terminated_length": 269.8953125, |
| "completions/min_length": 133.5, |
| "completions/min_terminated_length": 133.5, |
| "entropy": 0.11679979707114399, |
| "epoch": 0.4068522483940043, |
| "frac_reward_zero_std": 0.71875, |
| "grad_norm": 0.291015625, |
| "learning_rate": 9.981100000000002e-06, |
| "loss": 0.0018, |
| "num_tokens": 11264646.0, |
| "reward": 1.013203150033951, |
| "reward_std": 0.27041456326842306, |
| "rewards/reward_accuracy/mean": 0.91328125, |
| "rewards/reward_accuracy/std": 0.2702100805938244, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.956871461868286, |
| "sampling/importance_sampling_ratio/mean": 0.989966893196106, |
| "sampling/importance_sampling_ratio/min": 0.5115917325019836, |
| "sampling/sampling_logp_difference/max": 0.31674909591674805, |
| "sampling/sampling_logp_difference/mean": 0.003076907177455723, |
| "step": 190, |
| "step_time": 9.200360420020298 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 693.1, |
| "completions/max_terminated_length": 693.1, |
| "completions/mean_length": 269.8640625, |
| "completions/mean_terminated_length": 269.8640625, |
| "completions/min_length": 138.4, |
| "completions/min_terminated_length": 138.4, |
| "entropy": 0.11223944420926273, |
| "epoch": 0.4282655246252677, |
| "frac_reward_zero_std": 0.7625, |
| "grad_norm": 0.1201171875, |
| "learning_rate": 9.9801e-06, |
| "loss": -0.0023, |
| "num_tokens": 11869952.0, |
| "reward": 0.9984375238418579, |
| "reward_std": 0.27849631309509276, |
| "rewards/reward_accuracy/mean": 0.8984375, |
| "rewards/reward_accuracy/std": 0.27849631309509276, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.812314760684967, |
| "sampling/importance_sampling_ratio/mean": 0.999580317735672, |
| "sampling/importance_sampling_ratio/min": 0.4771026149392128, |
| "sampling/sampling_logp_difference/max": 0.2825710415840149, |
| "sampling/sampling_logp_difference/mean": 0.002931190375238657, |
| "step": 200, |
| "step_time": 8.994483379134909 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 666.4, |
| "completions/max_terminated_length": 666.4, |
| "completions/mean_length": 251.72109375, |
| "completions/mean_terminated_length": 251.72109375, |
| "completions/min_length": 131.1, |
| "completions/min_terminated_length": 131.1, |
| "entropy": 0.10416572629474104, |
| "epoch": 0.44967880085653106, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.1943359375, |
| "learning_rate": 9.9791e-06, |
| "loss": 0.0026, |
| "num_tokens": 12449627.0, |
| "reward": 1.0256250262260438, |
| "reward_std": 0.2459220364689827, |
| "rewards/reward_accuracy/mean": 0.92578125, |
| "rewards/reward_accuracy/std": 0.24549225419759751, |
| "rewards/reward_format/mean": 0.09984375238418579, |
| "rewards/reward_format/std": 0.001767767034471035, |
| "sampling/importance_sampling_ratio/max": 1.813460373878479, |
| "sampling/importance_sampling_ratio/mean": 1.0014194786548614, |
| "sampling/importance_sampling_ratio/min": 0.46973457038402555, |
| "sampling/sampling_logp_difference/max": 0.3183692216873169, |
| "sampling/sampling_logp_difference/mean": 0.002821452566422522, |
| "step": 210, |
| "step_time": 8.67334822844714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 535.8, |
| "completions/max_terminated_length": 535.8, |
| "completions/mean_length": 246.571875, |
| "completions/mean_terminated_length": 246.571875, |
| "completions/min_length": 128.8, |
| "completions/min_terminated_length": 128.8, |
| "entropy": 0.09552873424254357, |
| "epoch": 0.47109207708779444, |
| "frac_reward_zero_std": 0.80625, |
| "grad_norm": 0.1953125, |
| "learning_rate": 9.9781e-06, |
| "loss": -0.0014, |
| "num_tokens": 13025063.0, |
| "reward": 1.0187500238418579, |
| "reward_std": 0.26470365673303603, |
| "rewards/reward_accuracy/mean": 0.91875, |
| "rewards/reward_accuracy/std": 0.26470365673303603, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7015236616134644, |
| "sampling/importance_sampling_ratio/mean": 1.0000466644763946, |
| "sampling/importance_sampling_ratio/min": 0.5765772759914398, |
| "sampling/sampling_logp_difference/max": 0.32707799673080445, |
| "sampling/sampling_logp_difference/mean": 0.0026863845763728023, |
| "step": 220, |
| "step_time": 7.556477191485465 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 660.9, |
| "completions/max_terminated_length": 660.9, |
| "completions/mean_length": 255.27421875, |
| "completions/mean_terminated_length": 255.27421875, |
| "completions/min_length": 126.6, |
| "completions/min_terminated_length": 126.6, |
| "entropy": 0.10477679860778152, |
| "epoch": 0.4925053533190578, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.2080078125, |
| "learning_rate": 9.977100000000001e-06, |
| "loss": -0.0043, |
| "num_tokens": 13612134.0, |
| "reward": 1.012500023841858, |
| "reward_std": 0.277733413875103, |
| "rewards/reward_accuracy/mean": 0.9125, |
| "rewards/reward_accuracy/std": 0.277733413875103, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.697918176651001, |
| "sampling/importance_sampling_ratio/mean": 0.9932273745536804, |
| "sampling/importance_sampling_ratio/min": 0.46929139494895933, |
| "sampling/sampling_logp_difference/max": 0.3442947268486023, |
| "sampling/sampling_logp_difference/mean": 0.002886724704876542, |
| "step": 230, |
| "step_time": 8.76595054063946 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 536.1, |
| "completions/max_terminated_length": 536.1, |
| "completions/mean_length": 242.2515625, |
| "completions/mean_terminated_length": 242.2515625, |
| "completions/min_length": 125.2, |
| "completions/min_terminated_length": 125.2, |
| "entropy": 0.10415350631810724, |
| "epoch": 0.5139186295503212, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.2197265625, |
| "learning_rate": 9.976100000000001e-06, |
| "loss": -0.0043, |
| "num_tokens": 14187680.0, |
| "reward": 1.025781273841858, |
| "reward_std": 0.2497633121907711, |
| "rewards/reward_accuracy/mean": 0.92578125, |
| "rewards/reward_accuracy/std": 0.2497633121907711, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.775583016872406, |
| "sampling/importance_sampling_ratio/mean": 0.9898578584194183, |
| "sampling/importance_sampling_ratio/min": 0.5464063316583634, |
| "sampling/sampling_logp_difference/max": 0.3020105481147766, |
| "sampling/sampling_logp_difference/mean": 0.002806449821218848, |
| "step": 240, |
| "step_time": 7.506366596324369 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 578.1, |
| "completions/max_terminated_length": 578.1, |
| "completions/mean_length": 244.40078125, |
| "completions/mean_terminated_length": 244.40078125, |
| "completions/min_length": 124.2, |
| "completions/min_terminated_length": 124.2, |
| "entropy": 0.10175251732580363, |
| "epoch": 0.5353319057815846, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.2578125, |
| "learning_rate": 9.9751e-06, |
| "loss": 0.0007, |
| "num_tokens": 14758537.0, |
| "reward": 1.0039062738418578, |
| "reward_std": 0.28177610486745835, |
| "rewards/reward_accuracy/mean": 0.90390625, |
| "rewards/reward_accuracy/std": 0.28177610486745835, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8112457036972045, |
| "sampling/importance_sampling_ratio/mean": 0.9958719491958619, |
| "sampling/importance_sampling_ratio/min": 0.49983398616313934, |
| "sampling/sampling_logp_difference/max": 0.29114058017730715, |
| "sampling/sampling_logp_difference/mean": 0.002734354604035616, |
| "step": 250, |
| "step_time": 8.058183674840256 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 560.4, |
| "completions/max_terminated_length": 560.4, |
| "completions/mean_length": 244.940625, |
| "completions/mean_terminated_length": 244.940625, |
| "completions/min_length": 124.0, |
| "completions/min_terminated_length": 124.0, |
| "entropy": 0.1019574589561671, |
| "epoch": 0.556745182012848, |
| "frac_reward_zero_std": 0.7375, |
| "grad_norm": 0.44140625, |
| "learning_rate": 9.974100000000002e-06, |
| "loss": -0.0005, |
| "num_tokens": 15332381.0, |
| "reward": 1.0124219059944153, |
| "reward_std": 0.2744916543364525, |
| "rewards/reward_accuracy/mean": 0.9125, |
| "rewards/reward_accuracy/std": 0.2741738885641098, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.7419631719589233, |
| "sampling/importance_sampling_ratio/mean": 0.9960456371307373, |
| "sampling/importance_sampling_ratio/min": 0.5956381916999817, |
| "sampling/sampling_logp_difference/max": 0.27703256607055665, |
| "sampling/sampling_logp_difference/mean": 0.0027384211774915458, |
| "step": 260, |
| "step_time": 7.648829867923633 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 566.6, |
| "completions/max_terminated_length": 566.6, |
| "completions/mean_length": 253.81171875, |
| "completions/mean_terminated_length": 253.81171875, |
| "completions/min_length": 125.0, |
| "completions/min_terminated_length": 125.0, |
| "entropy": 0.10580904199741781, |
| "epoch": 0.5781584582441114, |
| "frac_reward_zero_std": 0.7375, |
| "grad_norm": 0.220703125, |
| "learning_rate": 9.973100000000001e-06, |
| "loss": 0.0001, |
| "num_tokens": 15913868.0, |
| "reward": 0.9984375238418579, |
| "reward_std": 0.2860999181866646, |
| "rewards/reward_accuracy/mean": 0.8984375, |
| "rewards/reward_accuracy/std": 0.2860999181866646, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7582767248153686, |
| "sampling/importance_sampling_ratio/mean": 0.9945892632007599, |
| "sampling/importance_sampling_ratio/min": 0.4966943830251694, |
| "sampling/sampling_logp_difference/max": 0.27067784070968626, |
| "sampling/sampling_logp_difference/mean": 0.002845003711991012, |
| "step": 270, |
| "step_time": 7.916164027014747 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 627.2, |
| "completions/max_terminated_length": 627.2, |
| "completions/mean_length": 257.38125, |
| "completions/mean_terminated_length": 257.38125, |
| "completions/min_length": 127.6, |
| "completions/min_terminated_length": 127.6, |
| "entropy": 0.10351922861300408, |
| "epoch": 0.5995717344753747, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.12255859375, |
| "learning_rate": 9.9721e-06, |
| "loss": -0.0026, |
| "num_tokens": 16500564.0, |
| "reward": 1.0101172089576722, |
| "reward_std": 0.27049388363957405, |
| "rewards/reward_accuracy/mean": 0.91015625, |
| "rewards/reward_accuracy/std": 0.2704001940786839, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.8087687849998475, |
| "sampling/importance_sampling_ratio/mean": 1.0013393700122832, |
| "sampling/importance_sampling_ratio/min": 0.5406015276908874, |
| "sampling/sampling_logp_difference/max": 0.2746386468410492, |
| "sampling/sampling_logp_difference/mean": 0.002734352136030793, |
| "step": 280, |
| "step_time": 8.25437042703852 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 639.6, |
| "completions/max_terminated_length": 639.6, |
| "completions/mean_length": 261.8125, |
| "completions/mean_terminated_length": 261.8125, |
| "completions/min_length": 127.3, |
| "completions/min_terminated_length": 127.3, |
| "entropy": 0.10294752875342965, |
| "epoch": 0.6209850107066381, |
| "frac_reward_zero_std": 0.76875, |
| "grad_norm": 0.435546875, |
| "learning_rate": 9.971100000000002e-06, |
| "loss": 0.006, |
| "num_tokens": 17096836.0, |
| "reward": 1.0304687738418579, |
| "reward_std": 0.24221332296729087, |
| "rewards/reward_accuracy/mean": 0.93046875, |
| "rewards/reward_accuracy/std": 0.24221332296729087, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7249974131584167, |
| "sampling/importance_sampling_ratio/mean": 1.00150528550148, |
| "sampling/importance_sampling_ratio/min": 0.5196426063776016, |
| "sampling/sampling_logp_difference/max": 0.29733462929725646, |
| "sampling/sampling_logp_difference/mean": 0.0027719730511307716, |
| "step": 290, |
| "step_time": 8.641226032143459 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 682.9, |
| "completions/max_terminated_length": 682.9, |
| "completions/mean_length": 261.59921875, |
| "completions/mean_terminated_length": 261.59921875, |
| "completions/min_length": 138.0, |
| "completions/min_terminated_length": 138.0, |
| "entropy": 0.09921529074199498, |
| "epoch": 0.6423982869379015, |
| "frac_reward_zero_std": 0.75625, |
| "grad_norm": 0.28515625, |
| "learning_rate": 9.9701e-06, |
| "loss": -0.0028, |
| "num_tokens": 17687995.0, |
| "reward": 0.9984375238418579, |
| "reward_std": 0.2913888946175575, |
| "rewards/reward_accuracy/mean": 0.8984375, |
| "rewards/reward_accuracy/std": 0.2913888916373253, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7411981582641602, |
| "sampling/importance_sampling_ratio/mean": 0.9974148631095886, |
| "sampling/importance_sampling_ratio/min": 0.47560388445854185, |
| "sampling/sampling_logp_difference/max": 0.39045414626598357, |
| "sampling/sampling_logp_difference/mean": 0.0028039254248142242, |
| "step": 300, |
| "step_time": 8.760718028061092 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 511.7, |
| "completions/max_terminated_length": 511.7, |
| "completions/mean_length": 255.50390625, |
| "completions/mean_terminated_length": 255.50390625, |
| "completions/min_length": 136.0, |
| "completions/min_terminated_length": 136.0, |
| "entropy": 0.09977072798646987, |
| "epoch": 0.6638115631691649, |
| "frac_reward_zero_std": 0.79375, |
| "grad_norm": 0.3203125, |
| "learning_rate": 9.9691e-06, |
| "loss": 0.0009, |
| "num_tokens": 18273936.0, |
| "reward": 1.0304687738418579, |
| "reward_std": 0.24341565147042274, |
| "rewards/reward_accuracy/mean": 0.93046875, |
| "rewards/reward_accuracy/std": 0.24341565147042274, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8566751718521117, |
| "sampling/importance_sampling_ratio/mean": 1.0085935950279237, |
| "sampling/importance_sampling_ratio/min": 0.5066721498966217, |
| "sampling/sampling_logp_difference/max": 0.29905726313591, |
| "sampling/sampling_logp_difference/mean": 0.0028781625675037502, |
| "step": 310, |
| "step_time": 7.511569580622018 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 645.4, |
| "completions/max_terminated_length": 645.4, |
| "completions/mean_length": 265.578125, |
| "completions/mean_terminated_length": 265.578125, |
| "completions/min_length": 143.8, |
| "completions/min_terminated_length": 143.8, |
| "entropy": 0.09197649753186851, |
| "epoch": 0.6852248394004282, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.23046875, |
| "learning_rate": 9.9681e-06, |
| "loss": 0.0013, |
| "num_tokens": 18869324.0, |
| "reward": 1.037500023841858, |
| "reward_std": 0.22793698459863662, |
| "rewards/reward_accuracy/mean": 0.9375, |
| "rewards/reward_accuracy/std": 0.22793698459863662, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7622220396995545, |
| "sampling/importance_sampling_ratio/mean": 1.0000321328639985, |
| "sampling/importance_sampling_ratio/min": 0.49555942714214324, |
| "sampling/sampling_logp_difference/max": 0.31999781131744387, |
| "sampling/sampling_logp_difference/mean": 0.002708026487380266, |
| "step": 320, |
| "step_time": 8.69271747865714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 626.0, |
| "completions/max_terminated_length": 626.0, |
| "completions/mean_length": 252.64609375, |
| "completions/mean_terminated_length": 252.64609375, |
| "completions/min_length": 134.2, |
| "completions/min_terminated_length": 134.2, |
| "entropy": 0.09282313482835888, |
| "epoch": 0.7066381156316917, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.265625, |
| "learning_rate": 9.9671e-06, |
| "loss": -0.0024, |
| "num_tokens": 19450143.0, |
| "reward": 1.0070312738418579, |
| "reward_std": 0.2594578005373478, |
| "rewards/reward_accuracy/mean": 0.90703125, |
| "rewards/reward_accuracy/std": 0.2594578005373478, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7192464470863342, |
| "sampling/importance_sampling_ratio/mean": 0.9964033842086792, |
| "sampling/importance_sampling_ratio/min": 0.5127495259046555, |
| "sampling/sampling_logp_difference/max": 0.3054734081029892, |
| "sampling/sampling_logp_difference/mean": 0.0027174823451787235, |
| "step": 330, |
| "step_time": 8.336150246160106 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 600.4, |
| "completions/max_terminated_length": 600.4, |
| "completions/mean_length": 247.384375, |
| "completions/mean_terminated_length": 247.384375, |
| "completions/min_length": 125.4, |
| "completions/min_terminated_length": 125.4, |
| "entropy": 0.088349405862391, |
| "epoch": 0.728051391862955, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.26171875, |
| "learning_rate": 9.966100000000001e-06, |
| "loss": -0.0024, |
| "num_tokens": 20025171.0, |
| "reward": 1.0257031559944152, |
| "reward_std": 0.2342689886689186, |
| "rewards/reward_accuracy/mean": 0.92578125, |
| "rewards/reward_accuracy/std": 0.2336314044892788, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.8207484126091003, |
| "sampling/importance_sampling_ratio/mean": 1.000370192527771, |
| "sampling/importance_sampling_ratio/min": 0.5644212096929551, |
| "sampling/sampling_logp_difference/max": 0.31171283721923826, |
| "sampling/sampling_logp_difference/mean": 0.0025741276098415256, |
| "step": 340, |
| "step_time": 8.245468164654449 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 627.6, |
| "completions/max_terminated_length": 627.6, |
| "completions/mean_length": 256.6078125, |
| "completions/mean_terminated_length": 256.6078125, |
| "completions/min_length": 130.1, |
| "completions/min_terminated_length": 130.1, |
| "entropy": 0.08159102343488485, |
| "epoch": 0.7494646680942184, |
| "frac_reward_zero_std": 0.76875, |
| "grad_norm": 0.1923828125, |
| "learning_rate": 9.9651e-06, |
| "loss": -0.0012, |
| "num_tokens": 20610877.0, |
| "reward": 0.9593750238418579, |
| "reward_std": 0.3399906471371651, |
| "rewards/reward_accuracy/mean": 0.859375, |
| "rewards/reward_accuracy/std": 0.3399906471371651, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8094131231307984, |
| "sampling/importance_sampling_ratio/mean": 0.9951273739337921, |
| "sampling/importance_sampling_ratio/min": 0.511447112262249, |
| "sampling/sampling_logp_difference/max": 0.44989765882492067, |
| "sampling/sampling_logp_difference/mean": 0.0024551642592996357, |
| "step": 350, |
| "step_time": 8.154948754329235 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 520.7, |
| "completions/max_terminated_length": 520.7, |
| "completions/mean_length": 253.171875, |
| "completions/mean_terminated_length": 253.171875, |
| "completions/min_length": 132.2, |
| "completions/min_terminated_length": 132.2, |
| "entropy": 0.08130336157046258, |
| "epoch": 0.7708779443254818, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.291015625, |
| "learning_rate": 9.964100000000002e-06, |
| "loss": -0.0001, |
| "num_tokens": 21193161.0, |
| "reward": 1.024218773841858, |
| "reward_std": 0.2590682491660118, |
| "rewards/reward_accuracy/mean": 0.92421875, |
| "rewards/reward_accuracy/std": 0.2590682491660118, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6712692618370055, |
| "sampling/importance_sampling_ratio/mean": 0.9935104370117187, |
| "sampling/importance_sampling_ratio/min": 0.5215151757001877, |
| "sampling/sampling_logp_difference/max": 0.29942506551742554, |
| "sampling/sampling_logp_difference/mean": 0.002472072094678879, |
| "step": 360, |
| "step_time": 7.57601957465522 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 672.9, |
| "completions/max_terminated_length": 672.9, |
| "completions/mean_length": 263.35234375, |
| "completions/mean_terminated_length": 263.35234375, |
| "completions/min_length": 132.7, |
| "completions/min_terminated_length": 132.7, |
| "entropy": 0.0922357952222228, |
| "epoch": 0.7922912205567452, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.33203125, |
| "learning_rate": 9.963100000000001e-06, |
| "loss": 0.0009, |
| "num_tokens": 21792692.0, |
| "reward": 1.028906273841858, |
| "reward_std": 0.23853662610054016, |
| "rewards/reward_accuracy/mean": 0.92890625, |
| "rewards/reward_accuracy/std": 0.23853662610054016, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7981587409973145, |
| "sampling/importance_sampling_ratio/mean": 0.9995792269706726, |
| "sampling/importance_sampling_ratio/min": 0.5026651382446289, |
| "sampling/sampling_logp_difference/max": 0.29441762566566465, |
| "sampling/sampling_logp_difference/mean": 0.0028194428654387594, |
| "step": 370, |
| "step_time": 8.88555830246769 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 698.9, |
| "completions/max_terminated_length": 698.9, |
| "completions/mean_length": 279.4765625, |
| "completions/mean_terminated_length": 279.4765625, |
| "completions/min_length": 142.9, |
| "completions/min_terminated_length": 142.9, |
| "entropy": 0.10615336340852081, |
| "epoch": 0.8137044967880086, |
| "frac_reward_zero_std": 0.79375, |
| "grad_norm": 0.2890625, |
| "learning_rate": 9.9621e-06, |
| "loss": -0.003, |
| "num_tokens": 22411238.0, |
| "reward": 0.9997656583786011, |
| "reward_std": 0.2907982409000397, |
| "rewards/reward_accuracy/mean": 0.9, |
| "rewards/reward_accuracy/std": 0.2901088684797287, |
| "rewards/reward_format/mean": 0.09976562708616257, |
| "rewards/reward_format/std": 0.00212895255535841, |
| "sampling/importance_sampling_ratio/max": 1.8302502751350402, |
| "sampling/importance_sampling_ratio/mean": 0.989685845375061, |
| "sampling/importance_sampling_ratio/min": 0.46740066558122634, |
| "sampling/sampling_logp_difference/max": 0.33320298194885256, |
| "sampling/sampling_logp_difference/mean": 0.003031467995606363, |
| "step": 380, |
| "step_time": 9.008702793717385 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 586.3, |
| "completions/max_terminated_length": 586.3, |
| "completions/mean_length": 292.1265625, |
| "completions/mean_terminated_length": 292.1265625, |
| "completions/min_length": 155.5, |
| "completions/min_terminated_length": 155.5, |
| "entropy": 0.10763896210119128, |
| "epoch": 0.8351177730192719, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.283203125, |
| "learning_rate": 9.961100000000002e-06, |
| "loss": -0.0034, |
| "num_tokens": 23045872.0, |
| "reward": 1.0045312702655793, |
| "reward_std": 0.2755120672285557, |
| "rewards/reward_accuracy/mean": 0.9046875, |
| "rewards/reward_accuracy/std": 0.27530695497989655, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0017677669413387776, |
| "sampling/importance_sampling_ratio/max": 1.7956003189086913, |
| "sampling/importance_sampling_ratio/mean": 1.0057555258274078, |
| "sampling/importance_sampling_ratio/min": 0.4832131594419479, |
| "sampling/sampling_logp_difference/max": 0.3077471494674683, |
| "sampling/sampling_logp_difference/mean": 0.0030803055968135597, |
| "step": 390, |
| "step_time": 8.129020752571524 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 848.6, |
| "completions/max_terminated_length": 848.6, |
| "completions/mean_length": 305.25859375, |
| "completions/mean_terminated_length": 305.25859375, |
| "completions/min_length": 165.4, |
| "completions/min_terminated_length": 165.4, |
| "entropy": 0.10830368623137474, |
| "epoch": 0.8565310492505354, |
| "frac_reward_zero_std": 0.75625, |
| "grad_norm": 0.1650390625, |
| "learning_rate": 9.9601e-06, |
| "loss": -0.0114, |
| "num_tokens": 23696387.0, |
| "reward": 1.0038281559944153, |
| "reward_std": 0.27770426496863365, |
| "rewards/reward_accuracy/mean": 0.90390625, |
| "rewards/reward_accuracy/std": 0.2777165465056896, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006225345656275749, |
| "sampling/importance_sampling_ratio/max": 1.8401121735572814, |
| "sampling/importance_sampling_ratio/mean": 0.9977504730224609, |
| "sampling/importance_sampling_ratio/min": 0.42685707062482836, |
| "sampling/sampling_logp_difference/max": 0.326308810710907, |
| "sampling/sampling_logp_difference/mean": 0.0031526062870398165, |
| "step": 400, |
| "step_time": 10.36821581046097 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 839.8, |
| "completions/max_terminated_length": 839.8, |
| "completions/mean_length": 303.12421875, |
| "completions/mean_terminated_length": 303.12421875, |
| "completions/min_length": 167.5, |
| "completions/min_terminated_length": 167.5, |
| "entropy": 0.10258147083222866, |
| "epoch": 0.8779443254817987, |
| "frac_reward_zero_std": 0.80625, |
| "grad_norm": 0.2041015625, |
| "learning_rate": 9.959100000000001e-06, |
| "loss": 0.0024, |
| "num_tokens": 24342410.0, |
| "reward": 1.0264062762260437, |
| "reward_std": 0.24076904952526093, |
| "rewards/reward_accuracy/mean": 0.9265625, |
| "rewards/reward_accuracy/std": 0.24066731631755828, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.001506417989730835, |
| "sampling/importance_sampling_ratio/max": 1.7575426697731018, |
| "sampling/importance_sampling_ratio/mean": 1.0029369354248048, |
| "sampling/importance_sampling_ratio/min": 0.5111173212528228, |
| "sampling/sampling_logp_difference/max": 0.32554699182510377, |
| "sampling/sampling_logp_difference/mean": 0.0030317373340949414, |
| "step": 410, |
| "step_time": 10.313068521488457 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 690.4, |
| "completions/max_terminated_length": 690.4, |
| "completions/mean_length": 297.53984375, |
| "completions/mean_terminated_length": 297.53984375, |
| "completions/min_length": 163.2, |
| "completions/min_terminated_length": 163.2, |
| "entropy": 0.09540976737625897, |
| "epoch": 0.8993576017130621, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.3125, |
| "learning_rate": 9.9581e-06, |
| "loss": -0.0049, |
| "num_tokens": 24979389.0, |
| "reward": 1.0280469059944153, |
| "reward_std": 0.2527216121554375, |
| "rewards/reward_accuracy/mean": 0.928125, |
| "rewards/reward_accuracy/std": 0.25243051499128344, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006225345656275749, |
| "sampling/importance_sampling_ratio/max": 1.931436264514923, |
| "sampling/importance_sampling_ratio/mean": 0.9963454127311706, |
| "sampling/importance_sampling_ratio/min": 0.46176286935806277, |
| "sampling/sampling_logp_difference/max": 0.2803131103515625, |
| "sampling/sampling_logp_difference/mean": 0.002849843422882259, |
| "step": 420, |
| "step_time": 8.845096204709261 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 674.9, |
| "completions/max_terminated_length": 674.9, |
| "completions/mean_length": 284.51328125, |
| "completions/mean_terminated_length": 284.51328125, |
| "completions/min_length": 159.1, |
| "completions/min_terminated_length": 159.1, |
| "entropy": 0.09750424907542765, |
| "epoch": 0.9207708779443254, |
| "frac_reward_zero_std": 0.74375, |
| "grad_norm": 0.1884765625, |
| "learning_rate": 9.9571e-06, |
| "loss": 0.0046, |
| "num_tokens": 25600318.0, |
| "reward": 0.989843773841858, |
| "reward_std": 0.2989785686135292, |
| "rewards/reward_accuracy/mean": 0.88984375, |
| "rewards/reward_accuracy/std": 0.2989785686135292, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9138119220733643, |
| "sampling/importance_sampling_ratio/mean": 0.998813945055008, |
| "sampling/importance_sampling_ratio/min": 0.46145528852939605, |
| "sampling/sampling_logp_difference/max": 0.3647011280059814, |
| "sampling/sampling_logp_difference/mean": 0.0029143803054466843, |
| "step": 430, |
| "step_time": 8.814448757423088 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 850.0, |
| "completions/max_terminated_length": 850.0, |
| "completions/mean_length": 278.81015625, |
| "completions/mean_terminated_length": 278.81015625, |
| "completions/min_length": 149.8, |
| "completions/min_terminated_length": 149.8, |
| "entropy": 0.09849745063111186, |
| "epoch": 0.9421841541755889, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.34765625, |
| "learning_rate": 9.956100000000001e-06, |
| "loss": 0.0005, |
| "num_tokens": 26214715.0, |
| "reward": 1.0429297089576721, |
| "reward_std": 0.20799603536725045, |
| "rewards/reward_accuracy/mean": 0.94296875, |
| "rewards/reward_accuracy/std": 0.20787103846669197, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.6965077757835387, |
| "sampling/importance_sampling_ratio/mean": 0.9814611375331879, |
| "sampling/importance_sampling_ratio/min": 0.4791929930448532, |
| "sampling/sampling_logp_difference/max": 0.29079715013504026, |
| "sampling/sampling_logp_difference/mean": 0.002953073102980852, |
| "step": 440, |
| "step_time": 10.410807088995352 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 812.6, |
| "completions/max_terminated_length": 812.6, |
| "completions/mean_length": 300.50234375, |
| "completions/mean_terminated_length": 300.50234375, |
| "completions/min_length": 152.6, |
| "completions/min_terminated_length": 152.6, |
| "entropy": 0.09185478910803795, |
| "epoch": 0.9635974304068522, |
| "frac_reward_zero_std": 0.7375, |
| "grad_norm": 0.265625, |
| "learning_rate": 9.9551e-06, |
| "loss": -0.0035, |
| "num_tokens": 26861278.0, |
| "reward": 0.9858594000339508, |
| "reward_std": 0.2981353662908077, |
| "rewards/reward_accuracy/mean": 0.8859375, |
| "rewards/reward_accuracy/std": 0.29799809083342554, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006225345656275749, |
| "sampling/importance_sampling_ratio/max": 1.9248536109924317, |
| "sampling/importance_sampling_ratio/mean": 1.010124933719635, |
| "sampling/importance_sampling_ratio/min": 0.401046958565712, |
| "sampling/sampling_logp_difference/max": 0.36202251017093656, |
| "sampling/sampling_logp_difference/mean": 0.0029199984157457946, |
| "step": 450, |
| "step_time": 9.870098124817014 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 981.6, |
| "completions/max_terminated_length": 981.6, |
| "completions/mean_length": 293.16953125, |
| "completions/mean_terminated_length": 293.16953125, |
| "completions/min_length": 159.3, |
| "completions/min_terminated_length": 159.3, |
| "entropy": 0.08230417687445879, |
| "epoch": 0.9850107066381156, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.193359375, |
| "learning_rate": 9.954100000000002e-06, |
| "loss": 0.0012, |
| "num_tokens": 27494911.0, |
| "reward": 1.0123828411102296, |
| "reward_std": 0.2602118641138077, |
| "rewards/reward_accuracy/mean": 0.9125, |
| "rewards/reward_accuracy/std": 0.2597881853580475, |
| "rewards/reward_format/mean": 0.09988281428813935, |
| "rewards/reward_format/std": 0.0013258252292871475, |
| "sampling/importance_sampling_ratio/max": 1.772279644012451, |
| "sampling/importance_sampling_ratio/mean": 0.9916043102741241, |
| "sampling/importance_sampling_ratio/min": 0.37843948900699614, |
| "sampling/sampling_logp_difference/max": 0.36489843130111693, |
| "sampling/sampling_logp_difference/mean": 0.0028459349879994987, |
| "step": 460, |
| "step_time": 11.615649558510631 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 670.7, |
| "completions/max_terminated_length": 670.7, |
| "completions/mean_length": 283.93046875, |
| "completions/mean_terminated_length": 283.93046875, |
| "completions/min_length": 167.6, |
| "completions/min_terminated_length": 167.6, |
| "entropy": 0.07738794998731464, |
| "epoch": 1.006423982869379, |
| "frac_reward_zero_std": 0.775, |
| "grad_norm": 0.4296875, |
| "learning_rate": 9.953100000000001e-06, |
| "loss": -0.0055, |
| "num_tokens": 28115326.0, |
| "reward": 1.0226562738418579, |
| "reward_std": 0.24386910125613212, |
| "rewards/reward_accuracy/mean": 0.92265625, |
| "rewards/reward_accuracy/std": 0.24386910125613212, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9326868891716003, |
| "sampling/importance_sampling_ratio/mean": 0.9995594501495362, |
| "sampling/importance_sampling_ratio/min": 0.4432160839438438, |
| "sampling/sampling_logp_difference/max": 0.556141710281372, |
| "sampling/sampling_logp_difference/mean": 0.0028032359201461076, |
| "step": 470, |
| "step_time": 8.692760161031037 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 659.4, |
| "completions/max_terminated_length": 659.4, |
| "completions/mean_length": 292.49453125, |
| "completions/mean_terminated_length": 292.49453125, |
| "completions/min_length": 170.7, |
| "completions/min_terminated_length": 170.7, |
| "entropy": 0.07248186687938869, |
| "epoch": 1.0278372591006424, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.29296875, |
| "learning_rate": 9.9521e-06, |
| "loss": -0.0031, |
| "num_tokens": 28749879.0, |
| "reward": 1.0560937881469727, |
| "reward_std": 0.19717978313565254, |
| "rewards/reward_accuracy/mean": 0.95625, |
| "rewards/reward_accuracy/std": 0.19631613418459892, |
| "rewards/reward_format/mean": 0.09984375238418579, |
| "rewards/reward_format/std": 0.001767767034471035, |
| "sampling/importance_sampling_ratio/max": 1.9483043193817138, |
| "sampling/importance_sampling_ratio/mean": 1.0084465205669404, |
| "sampling/importance_sampling_ratio/min": 0.46508778929710387, |
| "sampling/sampling_logp_difference/max": 0.3853133678436279, |
| "sampling/sampling_logp_difference/mean": 0.002750970027409494, |
| "step": 480, |
| "step_time": 8.76106935911812 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 857.0, |
| "completions/max_terminated_length": 857.0, |
| "completions/mean_length": 266.34453125, |
| "completions/mean_terminated_length": 266.34453125, |
| "completions/min_length": 142.7, |
| "completions/min_terminated_length": 142.7, |
| "entropy": 0.07258299100212753, |
| "epoch": 1.0492505353319057, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.25, |
| "learning_rate": 9.951100000000002e-06, |
| "loss": -0.0014, |
| "num_tokens": 29349656.0, |
| "reward": 1.0108594059944154, |
| "reward_std": 0.2591507524251938, |
| "rewards/reward_accuracy/mean": 0.9109375, |
| "rewards/reward_accuracy/std": 0.2588329896330833, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.7599544405937195, |
| "sampling/importance_sampling_ratio/mean": 0.9983604431152344, |
| "sampling/importance_sampling_ratio/min": 0.4861892879009247, |
| "sampling/sampling_logp_difference/max": 0.4126758337020874, |
| "sampling/sampling_logp_difference/mean": 0.002723811147734523, |
| "step": 490, |
| "step_time": 10.59247239441611 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 598.4, |
| "completions/max_terminated_length": 598.4, |
| "completions/mean_length": 254.32421875, |
| "completions/mean_terminated_length": 254.32421875, |
| "completions/min_length": 139.4, |
| "completions/min_terminated_length": 139.4, |
| "entropy": 0.0697967076441273, |
| "epoch": 1.0706638115631693, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.173828125, |
| "learning_rate": 9.9501e-06, |
| "loss": 0.0019, |
| "num_tokens": 29934183.0, |
| "reward": 1.0351562738418578, |
| "reward_std": 0.2357988677918911, |
| "rewards/reward_accuracy/mean": 0.93515625, |
| "rewards/reward_accuracy/std": 0.2357988677918911, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8208206057548524, |
| "sampling/importance_sampling_ratio/mean": 1.0105059146881104, |
| "sampling/importance_sampling_ratio/min": 0.45520285367965696, |
| "sampling/sampling_logp_difference/max": 0.34730725884437563, |
| "sampling/sampling_logp_difference/mean": 0.002632694412022829, |
| "step": 500, |
| "step_time": 7.992310962686315 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 635.7, |
| "completions/max_terminated_length": 635.7, |
| "completions/mean_length": 254.82109375, |
| "completions/mean_terminated_length": 254.82109375, |
| "completions/min_length": 137.5, |
| "completions/min_terminated_length": 137.5, |
| "entropy": 0.06628136513754726, |
| "epoch": 1.0920770877944326, |
| "frac_reward_zero_std": 0.9, |
| "grad_norm": 0.173828125, |
| "learning_rate": 9.949100000000001e-06, |
| "loss": -0.0004, |
| "num_tokens": 30522602.0, |
| "reward": 1.0578125238418579, |
| "reward_std": 0.18560680225491524, |
| "rewards/reward_accuracy/mean": 0.9578125, |
| "rewards/reward_accuracy/std": 0.18560680225491524, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7740282535552978, |
| "sampling/importance_sampling_ratio/mean": 0.9857339024543762, |
| "sampling/importance_sampling_ratio/min": 0.48116199374198915, |
| "sampling/sampling_logp_difference/max": 0.32534769773483274, |
| "sampling/sampling_logp_difference/mean": 0.0025019027292728425, |
| "step": 510, |
| "step_time": 8.63419182128273 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 653.1, |
| "completions/max_terminated_length": 653.1, |
| "completions/mean_length": 253.80546875, |
| "completions/mean_terminated_length": 253.80546875, |
| "completions/min_length": 134.2, |
| "completions/min_terminated_length": 134.2, |
| "entropy": 0.07210937552154065, |
| "epoch": 1.113490364025696, |
| "frac_reward_zero_std": 0.83125, |
| "grad_norm": 0.369140625, |
| "learning_rate": 9.9481e-06, |
| "loss": 0.0029, |
| "num_tokens": 31106049.0, |
| "reward": 1.045312523841858, |
| "reward_std": 0.20409346297383307, |
| "rewards/reward_accuracy/mean": 0.9453125, |
| "rewards/reward_accuracy/std": 0.20409346297383307, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7769490838050843, |
| "sampling/importance_sampling_ratio/mean": 1.0004009366035462, |
| "sampling/importance_sampling_ratio/min": 0.5122808903455734, |
| "sampling/sampling_logp_difference/max": 0.3511060833930969, |
| "sampling/sampling_logp_difference/mean": 0.0026746683986857535, |
| "step": 520, |
| "step_time": 8.520550536969676 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 572.2, |
| "completions/max_terminated_length": 572.2, |
| "completions/mean_length": 256.1359375, |
| "completions/mean_terminated_length": 256.1359375, |
| "completions/min_length": 142.4, |
| "completions/min_terminated_length": 142.4, |
| "entropy": 0.0703143150312826, |
| "epoch": 1.1349036402569592, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.27734375, |
| "learning_rate": 9.9471e-06, |
| "loss": -0.0033, |
| "num_tokens": 31694431.0, |
| "reward": 1.0312500238418578, |
| "reward_std": 0.2396256797015667, |
| "rewards/reward_accuracy/mean": 0.93125, |
| "rewards/reward_accuracy/std": 0.2396256797015667, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6923930764198303, |
| "sampling/importance_sampling_ratio/mean": 0.9930677771568298, |
| "sampling/importance_sampling_ratio/min": 0.47502332329750063, |
| "sampling/sampling_logp_difference/max": 0.3664711773395538, |
| "sampling/sampling_logp_difference/mean": 0.0025542341871187093, |
| "step": 530, |
| "step_time": 8.184974073944613 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 497.8, |
| "completions/max_terminated_length": 497.8, |
| "completions/mean_length": 256.67109375, |
| "completions/mean_terminated_length": 256.67109375, |
| "completions/min_length": 140.1, |
| "completions/min_terminated_length": 140.1, |
| "entropy": 0.06528630082029849, |
| "epoch": 1.1563169164882228, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.298828125, |
| "learning_rate": 9.946100000000001e-06, |
| "loss": -0.0003, |
| "num_tokens": 32281090.0, |
| "reward": 1.0569140911102295, |
| "reward_std": 0.1872013732790947, |
| "rewards/reward_accuracy/mean": 0.95703125, |
| "rewards/reward_accuracy/std": 0.1864813707768917, |
| "rewards/reward_format/mean": 0.09988281428813935, |
| "rewards/reward_format/std": 0.0013258252292871475, |
| "sampling/importance_sampling_ratio/max": 1.6891143798828125, |
| "sampling/importance_sampling_ratio/mean": 0.9918974220752717, |
| "sampling/importance_sampling_ratio/min": 0.5534315496683121, |
| "sampling/sampling_logp_difference/max": 0.3444695770740509, |
| "sampling/sampling_logp_difference/mean": 0.0023522253846749662, |
| "step": 540, |
| "step_time": 7.27368822558783 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 570.4, |
| "completions/max_terminated_length": 570.4, |
| "completions/mean_length": 260.44296875, |
| "completions/mean_terminated_length": 260.44296875, |
| "completions/min_length": 140.0, |
| "completions/min_terminated_length": 140.0, |
| "entropy": 0.06779216430149973, |
| "epoch": 1.177730192719486, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.244140625, |
| "learning_rate": 9.9451e-06, |
| "loss": -0.0006, |
| "num_tokens": 32872193.0, |
| "reward": 1.059335958957672, |
| "reward_std": 0.1830748811364174, |
| "rewards/reward_accuracy/mean": 0.959375, |
| "rewards/reward_accuracy/std": 0.18292889446020127, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.8134068727493287, |
| "sampling/importance_sampling_ratio/mean": 1.0031421601772308, |
| "sampling/importance_sampling_ratio/min": 0.516669774055481, |
| "sampling/sampling_logp_difference/max": 0.36979022026062014, |
| "sampling/sampling_logp_difference/mean": 0.0024435754399746656, |
| "step": 550, |
| "step_time": 7.866351552680134 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 543.8, |
| "completions/max_terminated_length": 543.8, |
| "completions/mean_length": 258.6296875, |
| "completions/mean_terminated_length": 258.6296875, |
| "completions/min_length": 138.9, |
| "completions/min_terminated_length": 138.9, |
| "entropy": 0.06997910283971578, |
| "epoch": 1.1991434689507494, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.1875, |
| "learning_rate": 9.9441e-06, |
| "loss": 0.0038, |
| "num_tokens": 33458919.0, |
| "reward": 1.047656273841858, |
| "reward_std": 0.20952882766723632, |
| "rewards/reward_accuracy/mean": 0.94765625, |
| "rewards/reward_accuracy/std": 0.20952882766723632, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7112900137901306, |
| "sampling/importance_sampling_ratio/mean": 0.9979656875133515, |
| "sampling/importance_sampling_ratio/min": 0.5579143524169922, |
| "sampling/sampling_logp_difference/max": 0.30935336351394654, |
| "sampling/sampling_logp_difference/mean": 0.002379806200042367, |
| "step": 560, |
| "step_time": 7.805754351941869 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 738.3, |
| "completions/max_terminated_length": 738.3, |
| "completions/mean_length": 286.38984375, |
| "completions/mean_terminated_length": 286.38984375, |
| "completions/min_length": 150.2, |
| "completions/min_terminated_length": 150.2, |
| "entropy": 0.07375946671236307, |
| "epoch": 1.2205567451820127, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.1572265625, |
| "learning_rate": 9.943100000000001e-06, |
| "loss": -0.0032, |
| "num_tokens": 34086362.0, |
| "reward": 1.021093773841858, |
| "reward_std": 0.2573475524783134, |
| "rewards/reward_accuracy/mean": 0.92109375, |
| "rewards/reward_accuracy/std": 0.2573475524783134, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9388673543930053, |
| "sampling/importance_sampling_ratio/mean": 0.9897044718265533, |
| "sampling/importance_sampling_ratio/min": 0.5029977947473526, |
| "sampling/sampling_logp_difference/max": 0.29439918994903563, |
| "sampling/sampling_logp_difference/mean": 0.0024436322040855885, |
| "step": 570, |
| "step_time": 9.273382845753805 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 841.0, |
| "completions/max_terminated_length": 841.0, |
| "completions/mean_length": 275.646875, |
| "completions/mean_terminated_length": 275.646875, |
| "completions/min_length": 139.0, |
| "completions/min_terminated_length": 139.0, |
| "entropy": 0.0781372680561617, |
| "epoch": 1.2419700214132763, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.50390625, |
| "learning_rate": 9.942100000000001e-06, |
| "loss": 0.0036, |
| "num_tokens": 34695550.0, |
| "reward": 1.016406273841858, |
| "reward_std": 0.2618931598961353, |
| "rewards/reward_accuracy/mean": 0.91640625, |
| "rewards/reward_accuracy/std": 0.2618931598961353, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8451479911804198, |
| "sampling/importance_sampling_ratio/mean": 0.9973420023918151, |
| "sampling/importance_sampling_ratio/min": 0.4617684125900269, |
| "sampling/sampling_logp_difference/max": 0.279401957988739, |
| "sampling/sampling_logp_difference/mean": 0.002499057212844491, |
| "step": 580, |
| "step_time": 10.610715378960595 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 620.5, |
| "completions/max_terminated_length": 620.5, |
| "completions/mean_length": 267.71015625, |
| "completions/mean_terminated_length": 267.71015625, |
| "completions/min_length": 138.8, |
| "completions/min_terminated_length": 138.8, |
| "entropy": 0.07485087802633643, |
| "epoch": 1.2633832976445396, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.1474609375, |
| "learning_rate": 9.941100000000002e-06, |
| "loss": -0.0057, |
| "num_tokens": 35294619.0, |
| "reward": 1.0397656500339507, |
| "reward_std": 0.21612160950899123, |
| "rewards/reward_accuracy/mean": 0.93984375, |
| "rewards/reward_accuracy/std": 0.21591712683439254, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.7671030521392823, |
| "sampling/importance_sampling_ratio/mean": 0.9916510045528412, |
| "sampling/importance_sampling_ratio/min": 0.5284463405609131, |
| "sampling/sampling_logp_difference/max": 0.31847329437732697, |
| "sampling/sampling_logp_difference/mean": 0.002557871420867741, |
| "step": 590, |
| "step_time": 8.362779534654692 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 805.2, |
| "completions/max_terminated_length": 805.2, |
| "completions/mean_length": 269.35078125, |
| "completions/mean_terminated_length": 269.35078125, |
| "completions/min_length": 144.4, |
| "completions/min_terminated_length": 144.4, |
| "entropy": 0.0702471622498706, |
| "epoch": 1.284796573875803, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.3359375, |
| "learning_rate": 9.9401e-06, |
| "loss": 0.0014, |
| "num_tokens": 35892140.0, |
| "reward": 1.021875023841858, |
| "reward_std": 0.2598454996943474, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.2598454996943474, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7072620511054992, |
| "sampling/importance_sampling_ratio/mean": 0.9866055548191071, |
| "sampling/importance_sampling_ratio/min": 0.43861359655857085, |
| "sampling/sampling_logp_difference/max": 0.402073347568512, |
| "sampling/sampling_logp_difference/mean": 0.002535216836258769, |
| "step": 600, |
| "step_time": 9.987932124035433 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 697.2, |
| "completions/max_terminated_length": 697.2, |
| "completions/mean_length": 272.6828125, |
| "completions/mean_terminated_length": 272.6828125, |
| "completions/min_length": 148.6, |
| "completions/min_terminated_length": 148.6, |
| "entropy": 0.06508733225055038, |
| "epoch": 1.3062098501070665, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.1572265625, |
| "learning_rate": 9.939100000000001e-06, |
| "loss": -0.0036, |
| "num_tokens": 36500366.0, |
| "reward": 1.0280468940734864, |
| "reward_std": 0.23932649046182633, |
| "rewards/reward_accuracy/mean": 0.928125, |
| "rewards/reward_accuracy/std": 0.23920153677463532, |
| "rewards/reward_format/mean": 0.09992187619209289, |
| "rewards/reward_format/std": 0.00088388342410326, |
| "sampling/importance_sampling_ratio/max": 1.7602363586425782, |
| "sampling/importance_sampling_ratio/mean": 0.9998273491859436, |
| "sampling/importance_sampling_ratio/min": 0.45725501179695127, |
| "sampling/sampling_logp_difference/max": 0.3096226751804352, |
| "sampling/sampling_logp_difference/mean": 0.0024219822604209184, |
| "step": 610, |
| "step_time": 8.984164946293458 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 617.6, |
| "completions/max_terminated_length": 617.6, |
| "completions/mean_length": 274.1390625, |
| "completions/mean_terminated_length": 274.1390625, |
| "completions/min_length": 145.3, |
| "completions/min_terminated_length": 145.3, |
| "entropy": 0.07019046759232879, |
| "epoch": 1.3276231263383298, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.1884765625, |
| "learning_rate": 9.9381e-06, |
| "loss": 0.0024, |
| "num_tokens": 37107392.0, |
| "reward": 1.060937523841858, |
| "reward_std": 0.1846742130815983, |
| "rewards/reward_accuracy/mean": 0.9609375, |
| "rewards/reward_accuracy/std": 0.1846742130815983, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8585228681564332, |
| "sampling/importance_sampling_ratio/mean": 1.0076585054397582, |
| "sampling/importance_sampling_ratio/min": 0.5134824931621551, |
| "sampling/sampling_logp_difference/max": 0.3875505208969116, |
| "sampling/sampling_logp_difference/mean": 0.0025750716216862203, |
| "step": 620, |
| "step_time": 8.374439669726417 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 708.9, |
| "completions/max_terminated_length": 708.9, |
| "completions/mean_length": 287.2375, |
| "completions/mean_terminated_length": 287.2375, |
| "completions/min_length": 145.7, |
| "completions/min_terminated_length": 145.7, |
| "entropy": 0.06904072989709675, |
| "epoch": 1.3490364025695931, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.083984375, |
| "learning_rate": 9.9371e-06, |
| "loss": 0.0003, |
| "num_tokens": 37734096.0, |
| "reward": 1.021875023841858, |
| "reward_std": 0.23618595078587531, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.23618595078587531, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7047074675559997, |
| "sampling/importance_sampling_ratio/mean": 1.0001424133777619, |
| "sampling/importance_sampling_ratio/min": 0.5024055182933808, |
| "sampling/sampling_logp_difference/max": 0.35920189023017884, |
| "sampling/sampling_logp_difference/mean": 0.002496272069402039, |
| "step": 630, |
| "step_time": 9.09644918604754 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 619.8, |
| "completions/max_terminated_length": 619.8, |
| "completions/mean_length": 277.8859375, |
| "completions/mean_terminated_length": 277.8859375, |
| "completions/min_length": 143.3, |
| "completions/min_terminated_length": 143.3, |
| "entropy": 0.0667171877110377, |
| "epoch": 1.3704496788008567, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.271484375, |
| "learning_rate": 9.936100000000001e-06, |
| "loss": 0.001, |
| "num_tokens": 38349398.0, |
| "reward": 1.0396875262260437, |
| "reward_std": 0.22937560379505156, |
| "rewards/reward_accuracy/mean": 0.93984375, |
| "rewards/reward_accuracy/std": 0.228865846991539, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012450690381228923, |
| "sampling/importance_sampling_ratio/max": 1.8871555566787719, |
| "sampling/importance_sampling_ratio/mean": 1.0114580869674683, |
| "sampling/importance_sampling_ratio/min": 0.4901000648736954, |
| "sampling/sampling_logp_difference/max": 0.3568188488483429, |
| "sampling/sampling_logp_difference/mean": 0.0024508767994120715, |
| "step": 640, |
| "step_time": 8.38586383536458 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 625.4, |
| "completions/max_terminated_length": 625.4, |
| "completions/mean_length": 283.25, |
| "completions/mean_terminated_length": 283.25, |
| "completions/min_length": 147.4, |
| "completions/min_terminated_length": 147.4, |
| "entropy": 0.06428178900387138, |
| "epoch": 1.39186295503212, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.1630859375, |
| "learning_rate": 9.9351e-06, |
| "loss": -0.0012, |
| "num_tokens": 38971910.0, |
| "reward": 1.0109375238418579, |
| "reward_std": 0.26264472082257273, |
| "rewards/reward_accuracy/mean": 0.9109375, |
| "rewards/reward_accuracy/std": 0.26264472082257273, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.757664954662323, |
| "sampling/importance_sampling_ratio/mean": 0.9957185566425324, |
| "sampling/importance_sampling_ratio/min": 0.46525874435901643, |
| "sampling/sampling_logp_difference/max": 0.3785123288631439, |
| "sampling/sampling_logp_difference/mean": 0.0024171099299564957, |
| "step": 650, |
| "step_time": 8.31470822719857 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 572.1, |
| "completions/max_terminated_length": 572.1, |
| "completions/mean_length": 269.7390625, |
| "completions/mean_terminated_length": 269.7390625, |
| "completions/min_length": 137.1, |
| "completions/min_terminated_length": 137.1, |
| "entropy": 0.06420559778343886, |
| "epoch": 1.4132762312633833, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 0.2431640625, |
| "learning_rate": 9.9341e-06, |
| "loss": -0.0028, |
| "num_tokens": 39573456.0, |
| "reward": 1.0421875238418579, |
| "reward_std": 0.21291813775897026, |
| "rewards/reward_accuracy/mean": 0.9421875, |
| "rewards/reward_accuracy/std": 0.21291813775897026, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.645384669303894, |
| "sampling/importance_sampling_ratio/mean": 0.9947456538677215, |
| "sampling/importance_sampling_ratio/min": 0.4917105585336685, |
| "sampling/sampling_logp_difference/max": 0.3389659285545349, |
| "sampling/sampling_logp_difference/mean": 0.002323517412878573, |
| "step": 660, |
| "step_time": 7.979478006483987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 679.5, |
| "completions/max_terminated_length": 679.5, |
| "completions/mean_length": 291.48671875, |
| "completions/mean_terminated_length": 291.48671875, |
| "completions/min_length": 149.8, |
| "completions/min_terminated_length": 149.8, |
| "entropy": 0.06897759772837161, |
| "epoch": 1.4346895074946466, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.3203125, |
| "learning_rate": 9.933100000000002e-06, |
| "loss": -0.0031, |
| "num_tokens": 40206687.0, |
| "reward": 1.0343750238418579, |
| "reward_std": 0.23865133672952651, |
| "rewards/reward_accuracy/mean": 0.934375, |
| "rewards/reward_accuracy/std": 0.23865133672952651, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.077260458469391, |
| "sampling/importance_sampling_ratio/mean": 1.0000692784786225, |
| "sampling/importance_sampling_ratio/min": 0.4843592494726181, |
| "sampling/sampling_logp_difference/max": 0.35375240445137024, |
| "sampling/sampling_logp_difference/mean": 0.002544402936473489, |
| "step": 670, |
| "step_time": 8.974567574355751 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 716.1, |
| "completions/max_terminated_length": 716.1, |
| "completions/mean_length": 298.71640625, |
| "completions/mean_terminated_length": 298.71640625, |
| "completions/min_length": 154.9, |
| "completions/min_terminated_length": 154.9, |
| "entropy": 0.07010278336238115, |
| "epoch": 1.45610278372591, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.2236328125, |
| "learning_rate": 9.932100000000001e-06, |
| "loss": -0.0015, |
| "num_tokens": 40850612.0, |
| "reward": 0.9951562762260437, |
| "reward_std": 0.2945816576480865, |
| "rewards/reward_accuracy/mean": 0.8953125, |
| "rewards/reward_accuracy/std": 0.294106251001358, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.000873381458222866, |
| "sampling/importance_sampling_ratio/max": 1.7501840472221375, |
| "sampling/importance_sampling_ratio/mean": 0.996564793586731, |
| "sampling/importance_sampling_ratio/min": 0.4214693635702133, |
| "sampling/sampling_logp_difference/max": 0.43126785159111025, |
| "sampling/sampling_logp_difference/mean": 0.0025064797140657903, |
| "step": 680, |
| "step_time": 9.038483161153271 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 611.0, |
| "completions/max_terminated_length": 611.0, |
| "completions/mean_length": 287.0046875, |
| "completions/mean_terminated_length": 287.0046875, |
| "completions/min_length": 155.9, |
| "completions/min_terminated_length": 155.9, |
| "entropy": 0.07116264782380313, |
| "epoch": 1.4775160599571735, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.197265625, |
| "learning_rate": 9.9311e-06, |
| "loss": 0.0004, |
| "num_tokens": 41475946.0, |
| "reward": 1.0319922149181366, |
| "reward_std": 0.22718643620610238, |
| "rewards/reward_accuracy/mean": 0.93203125, |
| "rewards/reward_accuracy/std": 0.2270868368446827, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.881392765045166, |
| "sampling/importance_sampling_ratio/mean": 0.9981233835220337, |
| "sampling/importance_sampling_ratio/min": 0.4819978952407837, |
| "sampling/sampling_logp_difference/max": 0.34613512754440307, |
| "sampling/sampling_logp_difference/mean": 0.002468063496053219, |
| "step": 690, |
| "step_time": 8.363545666681603 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 726.7, |
| "completions/max_terminated_length": 726.7, |
| "completions/mean_length": 279.62734375, |
| "completions/mean_terminated_length": 279.62734375, |
| "completions/min_length": 153.8, |
| "completions/min_terminated_length": 153.8, |
| "entropy": 0.07066833933349699, |
| "epoch": 1.4989293361884368, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.330078125, |
| "learning_rate": 9.9301e-06, |
| "loss": 0.0027, |
| "num_tokens": 42090237.0, |
| "reward": 1.041406273841858, |
| "reward_std": 0.2042766720056534, |
| "rewards/reward_accuracy/mean": 0.94140625, |
| "rewards/reward_accuracy/std": 0.2042766720056534, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.842268943786621, |
| "sampling/importance_sampling_ratio/mean": 1.0055402100086213, |
| "sampling/importance_sampling_ratio/min": 0.4462863504886627, |
| "sampling/sampling_logp_difference/max": 0.39222354292869566, |
| "sampling/sampling_logp_difference/mean": 0.002406470011919737, |
| "step": 700, |
| "step_time": 9.075394163327292 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 873.6, |
| "completions/max_terminated_length": 873.6, |
| "completions/mean_length": 304.75625, |
| "completions/mean_terminated_length": 304.75625, |
| "completions/min_length": 159.3, |
| "completions/min_terminated_length": 159.3, |
| "entropy": 0.07157183256931603, |
| "epoch": 1.5203426124197001, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.283203125, |
| "learning_rate": 9.929100000000001e-06, |
| "loss": 0.0009, |
| "num_tokens": 42742213.0, |
| "reward": 1.041406273841858, |
| "reward_std": 0.22278877571225167, |
| "rewards/reward_accuracy/mean": 0.94140625, |
| "rewards/reward_accuracy/std": 0.22278877571225167, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8749490976333618, |
| "sampling/importance_sampling_ratio/mean": 0.995981103181839, |
| "sampling/importance_sampling_ratio/min": 0.5061923563480377, |
| "sampling/sampling_logp_difference/max": 0.3489571034908295, |
| "sampling/sampling_logp_difference/mean": 0.002561335265636444, |
| "step": 710, |
| "step_time": 10.439321784162894 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 651.5, |
| "completions/max_terminated_length": 651.5, |
| "completions/mean_length": 304.534375, |
| "completions/mean_terminated_length": 304.534375, |
| "completions/min_length": 150.8, |
| "completions/min_terminated_length": 150.8, |
| "entropy": 0.06652877544984222, |
| "epoch": 1.5417558886509637, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.197265625, |
| "learning_rate": 9.9281e-06, |
| "loss": 0.0035, |
| "num_tokens": 43391377.0, |
| "reward": 1.0429687738418578, |
| "reward_std": 0.21550631001591683, |
| "rewards/reward_accuracy/mean": 0.94296875, |
| "rewards/reward_accuracy/std": 0.21550631001591683, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.851006770133972, |
| "sampling/importance_sampling_ratio/mean": 1.014693409204483, |
| "sampling/importance_sampling_ratio/min": 0.4772761255502701, |
| "sampling/sampling_logp_difference/max": 0.3900273621082306, |
| "sampling/sampling_logp_difference/mean": 0.002413895050995052, |
| "step": 720, |
| "step_time": 8.706314076762647 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 800.5, |
| "completions/max_terminated_length": 800.5, |
| "completions/mean_length": 321.7796875, |
| "completions/mean_terminated_length": 321.7796875, |
| "completions/min_length": 167.4, |
| "completions/min_terminated_length": 167.4, |
| "entropy": 0.06924776714295149, |
| "epoch": 1.563169164882227, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.146484375, |
| "learning_rate": 9.9271e-06, |
| "loss": -0.0054, |
| "num_tokens": 44069375.0, |
| "reward": 1.044492208957672, |
| "reward_std": 0.2050226479768753, |
| "rewards/reward_accuracy/mean": 0.94453125, |
| "rewards/reward_accuracy/std": 0.20485593378543854, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.7734390020370483, |
| "sampling/importance_sampling_ratio/mean": 0.9980226039886475, |
| "sampling/importance_sampling_ratio/min": 0.4823701769113541, |
| "sampling/sampling_logp_difference/max": 0.4007881969213486, |
| "sampling/sampling_logp_difference/mean": 0.0025082984706386925, |
| "step": 730, |
| "step_time": 9.875596550945193 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 691.7, |
| "completions/max_terminated_length": 691.7, |
| "completions/mean_length": 310.68359375, |
| "completions/mean_terminated_length": 310.68359375, |
| "completions/min_length": 173.0, |
| "completions/min_terminated_length": 173.0, |
| "entropy": 0.07169721641112119, |
| "epoch": 1.5845824411134903, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.0966796875, |
| "learning_rate": 9.926100000000001e-06, |
| "loss": 0.0004, |
| "num_tokens": 44723122.0, |
| "reward": 1.039843773841858, |
| "reward_std": 0.21003528982400893, |
| "rewards/reward_accuracy/mean": 0.93984375, |
| "rewards/reward_accuracy/std": 0.21003528982400893, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.816127610206604, |
| "sampling/importance_sampling_ratio/mean": 0.9968892157077789, |
| "sampling/importance_sampling_ratio/min": 0.4497199147939682, |
| "sampling/sampling_logp_difference/max": 0.3418192148208618, |
| "sampling/sampling_logp_difference/mean": 0.002535557607188821, |
| "step": 740, |
| "step_time": 8.952903557941317 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 733.2, |
| "completions/max_terminated_length": 733.2, |
| "completions/mean_length": 315.3828125, |
| "completions/mean_terminated_length": 315.3828125, |
| "completions/min_length": 173.4, |
| "completions/min_terminated_length": 173.4, |
| "entropy": 0.07107643098570407, |
| "epoch": 1.6059957173447539, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.158203125, |
| "learning_rate": 9.925100000000001e-06, |
| "loss": -0.0033, |
| "num_tokens": 45387524.0, |
| "reward": 1.001523458957672, |
| "reward_std": 0.2791046343743801, |
| "rewards/reward_accuracy/mean": 0.9015625, |
| "rewards/reward_accuracy/std": 0.2790135942399502, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.910252857208252, |
| "sampling/importance_sampling_ratio/mean": 0.9947206676006317, |
| "sampling/importance_sampling_ratio/min": 0.46861248314380644, |
| "sampling/sampling_logp_difference/max": 0.40812610387802123, |
| "sampling/sampling_logp_difference/mean": 0.0025402832543477416, |
| "step": 750, |
| "step_time": 9.246382595412433 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 766.4, |
| "completions/max_terminated_length": 766.4, |
| "completions/mean_length": 337.80703125, |
| "completions/mean_terminated_length": 337.80703125, |
| "completions/min_length": 175.7, |
| "completions/min_terminated_length": 175.7, |
| "entropy": 0.0686081362888217, |
| "epoch": 1.627408993576017, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.2353515625, |
| "learning_rate": 9.9241e-06, |
| "loss": -0.0054, |
| "num_tokens": 46084061.0, |
| "reward": 1.0038281500339508, |
| "reward_std": 0.27000613063573836, |
| "rewards/reward_accuracy/mean": 0.90390625, |
| "rewards/reward_accuracy/std": 0.2697835937142372, |
| "rewards/reward_format/mean": 0.09992187619209289, |
| "rewards/reward_format/std": 0.00088388342410326, |
| "sampling/importance_sampling_ratio/max": 2.0580177783966063, |
| "sampling/importance_sampling_ratio/mean": 0.9940634846687317, |
| "sampling/importance_sampling_ratio/min": 0.39175981730222703, |
| "sampling/sampling_logp_difference/max": 0.38021996021270754, |
| "sampling/sampling_logp_difference/mean": 0.0025734507013112306, |
| "step": 760, |
| "step_time": 9.401335881417618 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 751.3, |
| "completions/max_terminated_length": 751.3, |
| "completions/mean_length": 336.57265625, |
| "completions/mean_terminated_length": 336.57265625, |
| "completions/min_length": 190.5, |
| "completions/min_terminated_length": 190.5, |
| "entropy": 0.07014824904035777, |
| "epoch": 1.6488222698072805, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.322265625, |
| "learning_rate": 9.923100000000002e-06, |
| "loss": 0.0016, |
| "num_tokens": 46773634.0, |
| "reward": 1.0539062738418579, |
| "reward_std": 0.18094041720032691, |
| "rewards/reward_accuracy/mean": 0.95390625, |
| "rewards/reward_accuracy/std": 0.18094041720032691, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9697664022445678, |
| "sampling/importance_sampling_ratio/mean": 0.9891772747039795, |
| "sampling/importance_sampling_ratio/min": 0.38421621918678284, |
| "sampling/sampling_logp_difference/max": 0.47121219635009765, |
| "sampling/sampling_logp_difference/mean": 0.0026265155989676713, |
| "step": 770, |
| "step_time": 9.533575131185353 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 940.1, |
| "completions/max_terminated_length": 709.7, |
| "completions/mean_length": 351.89453125, |
| "completions/mean_terminated_length": 349.7677764892578, |
| "completions/min_length": 191.1, |
| "completions/min_terminated_length": 191.1, |
| "entropy": 0.06950077079236508, |
| "epoch": 1.6702355460385439, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 0.1552734375, |
| "learning_rate": 9.922100000000001e-06, |
| "loss": -0.0014, |
| "num_tokens": 47484539.0, |
| "reward": 1.044453150033951, |
| "reward_std": 0.20568459033966063, |
| "rewards/reward_accuracy/mean": 0.94453125, |
| "rewards/reward_accuracy/std": 0.20549212396144867, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 2.0638970494270326, |
| "sampling/importance_sampling_ratio/mean": 0.9883190870285035, |
| "sampling/importance_sampling_ratio/min": 0.43556567579507827, |
| "sampling/sampling_logp_difference/max": 0.40280606150627135, |
| "sampling/sampling_logp_difference/mean": 0.0025827130768448113, |
| "step": 780, |
| "step_time": 11.37751583456993 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 942.4, |
| "completions/max_terminated_length": 942.4, |
| "completions/mean_length": 347.1890625, |
| "completions/mean_terminated_length": 347.1890625, |
| "completions/min_length": 190.9, |
| "completions/min_terminated_length": 190.9, |
| "entropy": 0.07103997196536511, |
| "epoch": 1.6916488222698072, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.09912109375, |
| "learning_rate": 9.9211e-06, |
| "loss": -0.004, |
| "num_tokens": 48186101.0, |
| "reward": 1.031914097070694, |
| "reward_std": 0.23736095428466797, |
| "rewards/reward_accuracy/mean": 0.93203125, |
| "rewards/reward_accuracy/std": 0.2370909959077835, |
| "rewards/reward_format/mean": 0.09988281428813935, |
| "rewards/reward_format/std": 0.0013258252292871475, |
| "sampling/importance_sampling_ratio/max": 1.99387047290802, |
| "sampling/importance_sampling_ratio/mean": 1.0030252158641815, |
| "sampling/importance_sampling_ratio/min": 0.45869513154029845, |
| "sampling/sampling_logp_difference/max": 0.40425915718078614, |
| "sampling/sampling_logp_difference/mean": 0.0026623276760801675, |
| "step": 790, |
| "step_time": 11.129893393209205 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 669.5, |
| "completions/max_terminated_length": 669.5, |
| "completions/mean_length": 326.70390625, |
| "completions/mean_terminated_length": 326.70390625, |
| "completions/min_length": 188.0, |
| "completions/min_terminated_length": 188.0, |
| "entropy": 0.0701013681013137, |
| "epoch": 1.7130620985010707, |
| "frac_reward_zero_std": 0.86875, |
| "grad_norm": 0.298828125, |
| "learning_rate": 9.9201e-06, |
| "loss": -0.0031, |
| "num_tokens": 48860698.0, |
| "reward": 1.055429708957672, |
| "reward_std": 0.1734616495668888, |
| "rewards/reward_accuracy/mean": 0.95546875, |
| "rewards/reward_accuracy/std": 0.17346129640936853, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 2.0298019528388975, |
| "sampling/importance_sampling_ratio/mean": 1.0035560727119446, |
| "sampling/importance_sampling_ratio/min": 0.42651172578334806, |
| "sampling/sampling_logp_difference/max": 0.4085495352745056, |
| "sampling/sampling_logp_difference/mean": 0.0026044183177873492, |
| "step": 800, |
| "step_time": 8.965501307835803 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 762.5, |
| "completions/max_terminated_length": 762.5, |
| "completions/mean_length": 328.18125, |
| "completions/mean_terminated_length": 328.18125, |
| "completions/min_length": 186.0, |
| "completions/min_terminated_length": 186.0, |
| "entropy": 0.07027894344646483, |
| "epoch": 1.734475374732334, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.322265625, |
| "learning_rate": 9.9191e-06, |
| "loss": -0.0061, |
| "num_tokens": 49539346.0, |
| "reward": 1.0117187738418578, |
| "reward_std": 0.2709185376763344, |
| "rewards/reward_accuracy/mean": 0.91171875, |
| "rewards/reward_accuracy/std": 0.2709185376763344, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9254085183143617, |
| "sampling/importance_sampling_ratio/mean": 0.9959364712238312, |
| "sampling/importance_sampling_ratio/min": 0.42385431826114656, |
| "sampling/sampling_logp_difference/max": 0.3389554858207703, |
| "sampling/sampling_logp_difference/mean": 0.0026156638748943807, |
| "step": 810, |
| "step_time": 9.6866410497576 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 751.2, |
| "completions/max_terminated_length": 751.2, |
| "completions/mean_length": 335.4140625, |
| "completions/mean_terminated_length": 335.4140625, |
| "completions/min_length": 177.9, |
| "completions/min_terminated_length": 177.9, |
| "entropy": 0.07260147847700864, |
| "epoch": 1.7558886509635974, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.087890625, |
| "learning_rate": 9.9181e-06, |
| "loss": 0.0012, |
| "num_tokens": 50228556.0, |
| "reward": 1.0304687738418579, |
| "reward_std": 0.23274083510041238, |
| "rewards/reward_accuracy/mean": 0.93046875, |
| "rewards/reward_accuracy/std": 0.23274083510041238, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.794201385974884, |
| "sampling/importance_sampling_ratio/mean": 0.9981858968734741, |
| "sampling/importance_sampling_ratio/min": 0.4593837320804596, |
| "sampling/sampling_logp_difference/max": 0.4603695869445801, |
| "sampling/sampling_logp_difference/mean": 0.0026293183909729123, |
| "step": 820, |
| "step_time": 9.524095647735521 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 790.1, |
| "completions/max_terminated_length": 790.1, |
| "completions/mean_length": 325.04609375, |
| "completions/mean_terminated_length": 325.04609375, |
| "completions/min_length": 179.0, |
| "completions/min_terminated_length": 179.0, |
| "entropy": 0.0684969165828079, |
| "epoch": 1.777301927194861, |
| "frac_reward_zero_std": 0.83125, |
| "grad_norm": 0.1337890625, |
| "learning_rate": 9.9171e-06, |
| "loss": -0.001, |
| "num_tokens": 50903487.0, |
| "reward": 1.060937523841858, |
| "reward_std": 0.18465020507574081, |
| "rewards/reward_accuracy/mean": 0.9609375, |
| "rewards/reward_accuracy/std": 0.18465020507574081, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9602167725563049, |
| "sampling/importance_sampling_ratio/mean": 1.007552480697632, |
| "sampling/importance_sampling_ratio/min": 0.5122990429401397, |
| "sampling/sampling_logp_difference/max": 0.39109662771224973, |
| "sampling/sampling_logp_difference/mean": 0.0025083833374083043, |
| "step": 830, |
| "step_time": 10.03899248293601 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 769.8, |
| "completions/max_terminated_length": 769.8, |
| "completions/mean_length": 342.875, |
| "completions/mean_terminated_length": 342.875, |
| "completions/min_length": 196.5, |
| "completions/min_terminated_length": 196.5, |
| "entropy": 0.07339113471098244, |
| "epoch": 1.7987152034261242, |
| "frac_reward_zero_std": 0.80625, |
| "grad_norm": 0.267578125, |
| "learning_rate": 9.916100000000002e-06, |
| "loss": -0.0067, |
| "num_tokens": 51602047.0, |
| "reward": 1.017968773841858, |
| "reward_std": 0.2620562955737114, |
| "rewards/reward_accuracy/mean": 0.91796875, |
| "rewards/reward_accuracy/std": 0.2620562955737114, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.011053168773651, |
| "sampling/importance_sampling_ratio/mean": 1.0003466427326202, |
| "sampling/importance_sampling_ratio/min": 0.43487168103456497, |
| "sampling/sampling_logp_difference/max": 0.5305247724056243, |
| "sampling/sampling_logp_difference/mean": 0.0026388851227238776, |
| "step": 840, |
| "step_time": 9.436367471329868 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 733.5, |
| "completions/max_terminated_length": 733.5, |
| "completions/mean_length": 330.05859375, |
| "completions/mean_terminated_length": 330.05859375, |
| "completions/min_length": 176.0, |
| "completions/min_terminated_length": 176.0, |
| "entropy": 0.07355215607676654, |
| "epoch": 1.8201284796573876, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.30078125, |
| "learning_rate": 9.915100000000001e-06, |
| "loss": -0.0012, |
| "num_tokens": 52284682.0, |
| "reward": 1.0468359589576721, |
| "reward_std": 0.20284805223345756, |
| "rewards/reward_accuracy/mean": 0.946875, |
| "rewards/reward_accuracy/std": 0.20285320654511452, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 2.062855267524719, |
| "sampling/importance_sampling_ratio/mean": 0.9952085137367248, |
| "sampling/importance_sampling_ratio/min": 0.4312343098223209, |
| "sampling/sampling_logp_difference/max": 0.5963234424591064, |
| "sampling/sampling_logp_difference/mean": 0.0025386445922777057, |
| "step": 850, |
| "step_time": 9.429585955059157 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 703.0, |
| "completions/max_terminated_length": 703.0, |
| "completions/mean_length": 329.11171875, |
| "completions/mean_terminated_length": 329.11171875, |
| "completions/min_length": 180.8, |
| "completions/min_terminated_length": 180.8, |
| "entropy": 0.07494292494375258, |
| "epoch": 1.841541755888651, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.203125, |
| "learning_rate": 9.9141e-06, |
| "loss": -0.0045, |
| "num_tokens": 52965273.0, |
| "reward": 1.028906273841858, |
| "reward_std": 0.2468109667301178, |
| "rewards/reward_accuracy/mean": 0.92890625, |
| "rewards/reward_accuracy/std": 0.2468109667301178, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9343336582183839, |
| "sampling/importance_sampling_ratio/mean": 0.9980566620826721, |
| "sampling/importance_sampling_ratio/min": 0.4163599520921707, |
| "sampling/sampling_logp_difference/max": 0.32929205894470215, |
| "sampling/sampling_logp_difference/mean": 0.002617984754033387, |
| "step": 860, |
| "step_time": 8.97287191306241 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 1171.3, |
| "completions/max_terminated_length": 988.7, |
| "completions/mean_length": 338.7109375, |
| "completions/mean_terminated_length": 336.5625854492188, |
| "completions/min_length": 183.0, |
| "completions/min_terminated_length": 183.0, |
| "entropy": 0.0705749509157613, |
| "epoch": 1.8629550321199142, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.33984375, |
| "learning_rate": 9.913100000000002e-06, |
| "loss": -0.0035, |
| "num_tokens": 53658319.0, |
| "reward": 1.0326953411102295, |
| "reward_std": 0.2368771940469742, |
| "rewards/reward_accuracy/mean": 0.9328125, |
| "rewards/reward_accuracy/std": 0.23659040331840514, |
| "rewards/reward_format/mean": 0.09988281428813935, |
| "rewards/reward_format/std": 0.0013258252292871475, |
| "sampling/importance_sampling_ratio/max": 1.983928656578064, |
| "sampling/importance_sampling_ratio/mean": 0.9965438365936279, |
| "sampling/importance_sampling_ratio/min": 0.41132442504167555, |
| "sampling/sampling_logp_difference/max": 0.41971413493156434, |
| "sampling/sampling_logp_difference/mean": 0.002516378671862185, |
| "step": 870, |
| "step_time": 13.266158438380808 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 777.2, |
| "completions/max_terminated_length": 777.2, |
| "completions/mean_length": 316.1640625, |
| "completions/mean_terminated_length": 316.1640625, |
| "completions/min_length": 172.4, |
| "completions/min_terminated_length": 172.4, |
| "entropy": 0.07289869168307632, |
| "epoch": 1.8843683083511777, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.1796875, |
| "learning_rate": 9.912100000000001e-06, |
| "loss": -0.0018, |
| "num_tokens": 54320369.0, |
| "reward": 1.0273047089576721, |
| "reward_std": 0.23632968291640283, |
| "rewards/reward_accuracy/mean": 0.92734375, |
| "rewards/reward_accuracy/std": 0.23621006235480307, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.7956568002700806, |
| "sampling/importance_sampling_ratio/mean": 0.995242428779602, |
| "sampling/importance_sampling_ratio/min": 0.4729870170354843, |
| "sampling/sampling_logp_difference/max": 0.34785213470458987, |
| "sampling/sampling_logp_difference/mean": 0.002612321777269244, |
| "step": 880, |
| "step_time": 9.503890508506448 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 773.1, |
| "completions/max_terminated_length": 773.1, |
| "completions/mean_length": 305.4078125, |
| "completions/mean_terminated_length": 305.4078125, |
| "completions/min_length": 167.0, |
| "completions/min_terminated_length": 167.0, |
| "entropy": 0.07295166116673499, |
| "epoch": 1.905781584582441, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.1982421875, |
| "learning_rate": 9.9111e-06, |
| "loss": 0.0024, |
| "num_tokens": 54967851.0, |
| "reward": 1.0194922149181367, |
| "reward_std": 0.2340397544205189, |
| "rewards/reward_accuracy/mean": 0.91953125, |
| "rewards/reward_accuracy/std": 0.23405223861336708, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.7483533501625061, |
| "sampling/importance_sampling_ratio/mean": 1.0014864802360535, |
| "sampling/importance_sampling_ratio/min": 0.46369663178920745, |
| "sampling/sampling_logp_difference/max": 0.37404765486717223, |
| "sampling/sampling_logp_difference/mean": 0.0025562736205756663, |
| "step": 890, |
| "step_time": 9.566743845399468 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 614.9, |
| "completions/max_terminated_length": 614.9, |
| "completions/mean_length": 308.41171875, |
| "completions/mean_terminated_length": 308.41171875, |
| "completions/min_length": 161.6, |
| "completions/min_terminated_length": 161.6, |
| "entropy": 0.06940069403499365, |
| "epoch": 1.9271948608137044, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.1318359375, |
| "learning_rate": 9.9101e-06, |
| "loss": -0.0015, |
| "num_tokens": 55623762.0, |
| "reward": 1.0351562738418578, |
| "reward_std": 0.22706065103411674, |
| "rewards/reward_accuracy/mean": 0.93515625, |
| "rewards/reward_accuracy/std": 0.22706065103411674, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0028307914733885, |
| "sampling/importance_sampling_ratio/mean": 1.0073218882083892, |
| "sampling/importance_sampling_ratio/min": 0.518595427274704, |
| "sampling/sampling_logp_difference/max": 0.4238922715187073, |
| "sampling/sampling_logp_difference/mean": 0.002482188609428704, |
| "step": 900, |
| "step_time": 8.55545081733726 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 747.6, |
| "completions/max_terminated_length": 747.6, |
| "completions/mean_length": 302.20625, |
| "completions/mean_terminated_length": 302.20625, |
| "completions/min_length": 161.9, |
| "completions/min_terminated_length": 161.9, |
| "entropy": 0.06646617003716529, |
| "epoch": 1.948608137044968, |
| "frac_reward_zero_std": 0.86875, |
| "grad_norm": 0.1943359375, |
| "learning_rate": 9.9091e-06, |
| "loss": -0.0009, |
| "num_tokens": 56268722.0, |
| "reward": 1.0366406559944152, |
| "reward_std": 0.21777870506048203, |
| "rewards/reward_accuracy/mean": 0.93671875, |
| "rewards/reward_accuracy/std": 0.21777302399277687, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006225345656275749, |
| "sampling/importance_sampling_ratio/max": 1.9530771374702454, |
| "sampling/importance_sampling_ratio/mean": 1.002204167842865, |
| "sampling/importance_sampling_ratio/min": 0.46848204433918, |
| "sampling/sampling_logp_difference/max": 0.39162592887878417, |
| "sampling/sampling_logp_difference/mean": 0.0023698851000517607, |
| "step": 910, |
| "step_time": 9.52749267728068 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 689.9, |
| "completions/max_terminated_length": 689.9, |
| "completions/mean_length": 296.11953125, |
| "completions/mean_terminated_length": 296.11953125, |
| "completions/min_length": 159.1, |
| "completions/min_terminated_length": 159.1, |
| "entropy": 0.06819536592811345, |
| "epoch": 1.9700214132762313, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.1708984375, |
| "learning_rate": 9.908100000000001e-06, |
| "loss": -0.0038, |
| "num_tokens": 56906267.0, |
| "reward": 1.047656273841858, |
| "reward_std": 0.20231884270906447, |
| "rewards/reward_accuracy/mean": 0.94765625, |
| "rewards/reward_accuracy/std": 0.20231884270906447, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.967441737651825, |
| "sampling/importance_sampling_ratio/mean": 0.9984501481056214, |
| "sampling/importance_sampling_ratio/min": 0.4701683551073074, |
| "sampling/sampling_logp_difference/max": 0.4156764030456543, |
| "sampling/sampling_logp_difference/mean": 0.0024588283151388167, |
| "step": 920, |
| "step_time": 9.010744774807245 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 699.6, |
| "completions/max_terminated_length": 699.6, |
| "completions/mean_length": 300.2296875, |
| "completions/mean_terminated_length": 300.2296875, |
| "completions/min_length": 158.2, |
| "completions/min_terminated_length": 158.2, |
| "entropy": 0.06864205857273191, |
| "epoch": 1.9914346895074946, |
| "frac_reward_zero_std": 0.83125, |
| "grad_norm": 0.322265625, |
| "learning_rate": 9.9071e-06, |
| "loss": -0.0011, |
| "num_tokens": 57552569.0, |
| "reward": 0.9944922089576721, |
| "reward_std": 0.28288685381412504, |
| "rewards/reward_accuracy/mean": 0.89453125, |
| "rewards/reward_accuracy/std": 0.28281374275684357, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.7924079895019531, |
| "sampling/importance_sampling_ratio/mean": 1.0052897214889527, |
| "sampling/importance_sampling_ratio/min": 0.49464659094810487, |
| "sampling/sampling_logp_difference/max": 0.3956571102142334, |
| "sampling/sampling_logp_difference/mean": 0.0024114875588566063, |
| "step": 930, |
| "step_time": 8.920792492292822 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 605.6, |
| "completions/max_terminated_length": 605.6, |
| "completions/mean_length": 291.99296875, |
| "completions/mean_terminated_length": 291.99296875, |
| "completions/min_length": 153.4, |
| "completions/min_terminated_length": 153.4, |
| "entropy": 0.068846017238684, |
| "epoch": 2.012847965738758, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.1640625, |
| "learning_rate": 9.9061e-06, |
| "loss": 0.0014, |
| "num_tokens": 58180000.0, |
| "reward": 1.055429708957672, |
| "reward_std": 0.16549216210842133, |
| "rewards/reward_accuracy/mean": 0.95546875, |
| "rewards/reward_accuracy/std": 0.1654101938009262, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.7594813466072083, |
| "sampling/importance_sampling_ratio/mean": 0.9936805665493011, |
| "sampling/importance_sampling_ratio/min": 0.5412941426038742, |
| "sampling/sampling_logp_difference/max": 0.3487249732017517, |
| "sampling/sampling_logp_difference/mean": 0.0024198970291763543, |
| "step": 940, |
| "step_time": 8.233999243704602 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 644.6, |
| "completions/max_terminated_length": 644.6, |
| "completions/mean_length": 306.16171875, |
| "completions/mean_terminated_length": 306.16171875, |
| "completions/min_length": 157.0, |
| "completions/min_terminated_length": 157.0, |
| "entropy": 0.07199717878829688, |
| "epoch": 2.0342612419700212, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.201171875, |
| "learning_rate": 9.905100000000001e-06, |
| "loss": -0.0018, |
| "num_tokens": 58832151.0, |
| "reward": 1.043750023841858, |
| "reward_std": 0.21752603575587273, |
| "rewards/reward_accuracy/mean": 0.94375, |
| "rewards/reward_accuracy/std": 0.21752603575587273, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8271437883377075, |
| "sampling/importance_sampling_ratio/mean": 0.9923692941665649, |
| "sampling/importance_sampling_ratio/min": 0.45557867288589476, |
| "sampling/sampling_logp_difference/max": 0.33249701261520387, |
| "sampling/sampling_logp_difference/mean": 0.0024716480635106563, |
| "step": 950, |
| "step_time": 8.502925189770759 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 818.9, |
| "completions/max_terminated_length": 818.9, |
| "completions/mean_length": 315.78046875, |
| "completions/mean_terminated_length": 315.78046875, |
| "completions/min_length": 149.9, |
| "completions/min_terminated_length": 149.9, |
| "entropy": 0.0738558794837445, |
| "epoch": 2.0556745182012848, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.1337890625, |
| "learning_rate": 9.9041e-06, |
| "loss": 0.0034, |
| "num_tokens": 59495358.0, |
| "reward": 1.0390625238418578, |
| "reward_std": 0.22255490422248841, |
| "rewards/reward_accuracy/mean": 0.9390625, |
| "rewards/reward_accuracy/std": 0.22255490422248841, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.98711838722229, |
| "sampling/importance_sampling_ratio/mean": 1.0085735499858857, |
| "sampling/importance_sampling_ratio/min": 0.40030335187911986, |
| "sampling/sampling_logp_difference/max": 0.32944337129592893, |
| "sampling/sampling_logp_difference/mean": 0.0025197431445121766, |
| "step": 960, |
| "step_time": 9.975385126471519 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 637.8, |
| "completions/max_terminated_length": 637.8, |
| "completions/mean_length": 301.32265625, |
| "completions/mean_terminated_length": 301.32265625, |
| "completions/min_length": 157.3, |
| "completions/min_terminated_length": 157.3, |
| "entropy": 0.07220690306276083, |
| "epoch": 2.0770877944325483, |
| "frac_reward_zero_std": 0.90625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.903100000000002e-06, |
| "loss": -0.0001, |
| "num_tokens": 60141339.0, |
| "reward": 1.0663672089576721, |
| "reward_std": 0.13860948085784913, |
| "rewards/reward_accuracy/mean": 0.96640625, |
| "rewards/reward_accuracy/std": 0.13844276666641236, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.736439287662506, |
| "sampling/importance_sampling_ratio/mean": 1.0019245147705078, |
| "sampling/importance_sampling_ratio/min": 0.5247830837965012, |
| "sampling/sampling_logp_difference/max": 0.3802154064178467, |
| "sampling/sampling_logp_difference/mean": 0.0024656007997691633, |
| "step": 970, |
| "step_time": 8.521714383177459 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 860.8, |
| "completions/max_terminated_length": 860.8, |
| "completions/mean_length": 309.1265625, |
| "completions/mean_terminated_length": 309.1265625, |
| "completions/min_length": 173.3, |
| "completions/min_terminated_length": 173.3, |
| "entropy": 0.07528793059755116, |
| "epoch": 2.0985010706638114, |
| "frac_reward_zero_std": 0.83125, |
| "grad_norm": 0.296875, |
| "learning_rate": 9.902100000000001e-06, |
| "loss": -0.0069, |
| "num_tokens": 60795469.0, |
| "reward": 1.0234375238418578, |
| "reward_std": 0.25523324608802794, |
| "rewards/reward_accuracy/mean": 0.9234375, |
| "rewards/reward_accuracy/std": 0.25523324608802794, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7005436182022096, |
| "sampling/importance_sampling_ratio/mean": 0.9892678797245026, |
| "sampling/importance_sampling_ratio/min": 0.4936142176389694, |
| "sampling/sampling_logp_difference/max": 0.36187576651573183, |
| "sampling/sampling_logp_difference/mean": 0.0024184143636375665, |
| "step": 980, |
| "step_time": 10.34130103830248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 692.3, |
| "completions/max_terminated_length": 692.3, |
| "completions/mean_length": 312.7265625, |
| "completions/mean_terminated_length": 312.7265625, |
| "completions/min_length": 159.6, |
| "completions/min_terminated_length": 159.6, |
| "entropy": 0.0728087070165202, |
| "epoch": 2.119914346895075, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9011e-06, |
| "loss": -0.0013, |
| "num_tokens": 61455031.0, |
| "reward": 1.0468750238418578, |
| "reward_std": 0.20668290555477142, |
| "rewards/reward_accuracy/mean": 0.946875, |
| "rewards/reward_accuracy/std": 0.20668290555477142, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.845291757583618, |
| "sampling/importance_sampling_ratio/mean": 1.0025856792926788, |
| "sampling/importance_sampling_ratio/min": 0.42245843410491946, |
| "sampling/sampling_logp_difference/max": 0.33411277532577516, |
| "sampling/sampling_logp_difference/mean": 0.0023591268341988324, |
| "step": 990, |
| "step_time": 9.175962283695117 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 601.3, |
| "completions/max_terminated_length": 601.3, |
| "completions/mean_length": 313.06875, |
| "completions/mean_terminated_length": 313.06875, |
| "completions/min_length": 166.8, |
| "completions/min_terminated_length": 166.8, |
| "entropy": 0.07901076532434673, |
| "epoch": 2.1413276231263385, |
| "frac_reward_zero_std": 0.85625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.9001e-06, |
| "loss": 0.0017, |
| "num_tokens": 62114895.0, |
| "reward": 1.049218773841858, |
| "reward_std": 0.20678736120462418, |
| "rewards/reward_accuracy/mean": 0.94921875, |
| "rewards/reward_accuracy/std": 0.20678736120462418, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7237709760665894, |
| "sampling/importance_sampling_ratio/mean": 1.002291339635849, |
| "sampling/importance_sampling_ratio/min": 0.5337045043706894, |
| "sampling/sampling_logp_difference/max": 0.36232264041900636, |
| "sampling/sampling_logp_difference/mean": 0.002454755920916796, |
| "step": 1000, |
| "step_time": 8.217873241659253 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 847.8, |
| "completions/max_terminated_length": 847.8, |
| "completions/mean_length": 325.3, |
| "completions/mean_terminated_length": 325.3, |
| "completions/min_length": 160.5, |
| "completions/min_terminated_length": 160.5, |
| "entropy": 0.08362047534901648, |
| "epoch": 2.1627408993576016, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.30859375, |
| "learning_rate": 9.8991e-06, |
| "loss": 0.0005, |
| "num_tokens": 62792063.0, |
| "reward": 1.041406273841858, |
| "reward_std": 0.19614977464079858, |
| "rewards/reward_accuracy/mean": 0.94140625, |
| "rewards/reward_accuracy/std": 0.19614977464079858, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8516573548316955, |
| "sampling/importance_sampling_ratio/mean": 0.9924830496311188, |
| "sampling/importance_sampling_ratio/min": 0.40951538980007174, |
| "sampling/sampling_logp_difference/max": 0.31194607019424436, |
| "sampling/sampling_logp_difference/mean": 0.0025415489915758373, |
| "step": 1010, |
| "step_time": 10.452530771447346 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 584.0, |
| "completions/max_terminated_length": 584.0, |
| "completions/mean_length": 306.2640625, |
| "completions/mean_terminated_length": 306.2640625, |
| "completions/min_length": 161.8, |
| "completions/min_terminated_length": 161.8, |
| "entropy": 0.08374462318606675, |
| "epoch": 2.184154175588865, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.3125, |
| "learning_rate": 9.898100000000001e-06, |
| "loss": -0.0026, |
| "num_tokens": 63441841.0, |
| "reward": 1.0382812738418579, |
| "reward_std": 0.20318744108080863, |
| "rewards/reward_accuracy/mean": 0.93828125, |
| "rewards/reward_accuracy/std": 0.20318744108080863, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8438876271247864, |
| "sampling/importance_sampling_ratio/mean": 0.9989378809928894, |
| "sampling/importance_sampling_ratio/min": 0.4692339837551117, |
| "sampling/sampling_logp_difference/max": 0.4050160586833954, |
| "sampling/sampling_logp_difference/mean": 0.002553582424297929, |
| "step": 1020, |
| "step_time": 8.046002806304022 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 704.2, |
| "completions/max_terminated_length": 704.2, |
| "completions/mean_length": 299.6578125, |
| "completions/mean_terminated_length": 299.6578125, |
| "completions/min_length": 153.8, |
| "completions/min_terminated_length": 153.8, |
| "entropy": 0.08486975315026939, |
| "epoch": 2.2055674518201283, |
| "frac_reward_zero_std": 0.89375, |
| "grad_norm": 0.1787109375, |
| "learning_rate": 9.8971e-06, |
| "loss": -0.0033, |
| "num_tokens": 64079819.0, |
| "reward": 1.017968773841858, |
| "reward_std": 0.23034121170639993, |
| "rewards/reward_accuracy/mean": 0.91796875, |
| "rewards/reward_accuracy/std": 0.23034121170639993, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8897068738937377, |
| "sampling/importance_sampling_ratio/mean": 0.9994602501392365, |
| "sampling/importance_sampling_ratio/min": 0.4835642784833908, |
| "sampling/sampling_logp_difference/max": 0.3478668868541718, |
| "sampling/sampling_logp_difference/mean": 0.002564625651575625, |
| "step": 1030, |
| "step_time": 8.886243190104143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 668.9, |
| "completions/max_terminated_length": 668.9, |
| "completions/mean_length": 300.42421875, |
| "completions/mean_terminated_length": 300.42421875, |
| "completions/min_length": 166.1, |
| "completions/min_terminated_length": 166.1, |
| "entropy": 0.08346233125776052, |
| "epoch": 2.226980728051392, |
| "frac_reward_zero_std": 0.90625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8961e-06, |
| "loss": -0.0006, |
| "num_tokens": 64722306.0, |
| "reward": 1.047656273841858, |
| "reward_std": 0.2017066664993763, |
| "rewards/reward_accuracy/mean": 0.94765625, |
| "rewards/reward_accuracy/std": 0.2017066664993763, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7974767327308654, |
| "sampling/importance_sampling_ratio/mean": 1.0007626295089722, |
| "sampling/importance_sampling_ratio/min": 0.5009303316473961, |
| "sampling/sampling_logp_difference/max": 0.3322217702865601, |
| "sampling/sampling_logp_difference/mean": 0.0025520122377201914, |
| "step": 1040, |
| "step_time": 8.74352624192834 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 567.6, |
| "completions/max_terminated_length": 567.6, |
| "completions/mean_length": 293.2484375, |
| "completions/mean_terminated_length": 293.2484375, |
| "completions/min_length": 165.3, |
| "completions/min_terminated_length": 165.3, |
| "entropy": 0.08366065071895719, |
| "epoch": 2.2483940042826553, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.2431640625, |
| "learning_rate": 9.895100000000001e-06, |
| "loss": -0.0017, |
| "num_tokens": 65354792.0, |
| "reward": 1.0585937738418578, |
| "reward_std": 0.18354986757040023, |
| "rewards/reward_accuracy/mean": 0.95859375, |
| "rewards/reward_accuracy/std": 0.18354986757040023, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.787668192386627, |
| "sampling/importance_sampling_ratio/mean": 1.0064069867134093, |
| "sampling/importance_sampling_ratio/min": 0.49496580958366393, |
| "sampling/sampling_logp_difference/max": 0.2858777821063995, |
| "sampling/sampling_logp_difference/mean": 0.002530720317736268, |
| "step": 1050, |
| "step_time": 7.99899155665189 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 826.4, |
| "completions/max_terminated_length": 826.4, |
| "completions/mean_length": 324.5625, |
| "completions/mean_terminated_length": 324.5625, |
| "completions/min_length": 173.0, |
| "completions/min_terminated_length": 173.0, |
| "entropy": 0.09151477455161512, |
| "epoch": 2.2698072805139184, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.10986328125, |
| "learning_rate": 9.8941e-06, |
| "loss": 0.0014, |
| "num_tokens": 66030248.0, |
| "reward": 1.0491797149181366, |
| "reward_std": 0.18389897271990777, |
| "rewards/reward_accuracy/mean": 0.94921875, |
| "rewards/reward_accuracy/std": 0.18378850147128106, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.9361152529716492, |
| "sampling/importance_sampling_ratio/mean": 0.9886549592018128, |
| "sampling/importance_sampling_ratio/min": 0.43754925578832626, |
| "sampling/sampling_logp_difference/max": 0.3484856605529785, |
| "sampling/sampling_logp_difference/mean": 0.0027168720494955777, |
| "step": 1060, |
| "step_time": 10.222180938441307 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 703.5, |
| "completions/max_terminated_length": 703.5, |
| "completions/mean_length": 300.88984375, |
| "completions/mean_terminated_length": 300.88984375, |
| "completions/min_length": 154.9, |
| "completions/min_terminated_length": 154.9, |
| "entropy": 0.08877753522247075, |
| "epoch": 2.291220556745182, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.212890625, |
| "learning_rate": 9.8931e-06, |
| "loss": -0.0022, |
| "num_tokens": 66672707.0, |
| "reward": 1.0109375238418579, |
| "reward_std": 0.26612810492515565, |
| "rewards/reward_accuracy/mean": 0.9109375, |
| "rewards/reward_accuracy/std": 0.26612810492515565, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7829594969749452, |
| "sampling/importance_sampling_ratio/mean": 1.0007531762123107, |
| "sampling/importance_sampling_ratio/min": 0.5367873042821885, |
| "sampling/sampling_logp_difference/max": 0.3121935844421387, |
| "sampling/sampling_logp_difference/mean": 0.0027119210222736, |
| "step": 1070, |
| "step_time": 9.056643118103967 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 673.6, |
| "completions/max_terminated_length": 673.6, |
| "completions/mean_length": 310.9296875, |
| "completions/mean_terminated_length": 310.9296875, |
| "completions/min_length": 169.6, |
| "completions/min_terminated_length": 169.6, |
| "entropy": 0.08264521139208228, |
| "epoch": 2.3126338329764455, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.267578125, |
| "learning_rate": 9.892100000000001e-06, |
| "loss": 0.0033, |
| "num_tokens": 67330377.0, |
| "reward": 1.045312523841858, |
| "reward_std": 0.2131926476955414, |
| "rewards/reward_accuracy/mean": 0.9453125, |
| "rewards/reward_accuracy/std": 0.2131926476955414, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8953032493591309, |
| "sampling/importance_sampling_ratio/mean": 0.9981930792331696, |
| "sampling/importance_sampling_ratio/min": 0.47417056411504743, |
| "sampling/sampling_logp_difference/max": 0.45659106969833374, |
| "sampling/sampling_logp_difference/mean": 0.0026141755748540162, |
| "step": 1080, |
| "step_time": 8.860386438667774 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 595.8, |
| "completions/max_terminated_length": 595.8, |
| "completions/mean_length": 304.48671875, |
| "completions/mean_terminated_length": 304.48671875, |
| "completions/min_length": 156.0, |
| "completions/min_terminated_length": 156.0, |
| "entropy": 0.07733453779947012, |
| "epoch": 2.3340471092077086, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.891100000000001e-06, |
| "loss": -0.0019, |
| "num_tokens": 67977664.0, |
| "reward": 1.0569922089576722, |
| "reward_std": 0.18417541831731796, |
| "rewards/reward_accuracy/mean": 0.95703125, |
| "rewards/reward_accuracy/std": 0.18418469578027724, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.6912529468536377, |
| "sampling/importance_sampling_ratio/mean": 0.9908089995384216, |
| "sampling/importance_sampling_ratio/min": 0.46684759557247163, |
| "sampling/sampling_logp_difference/max": 0.3770546555519104, |
| "sampling/sampling_logp_difference/mean": 0.0024958117865025997, |
| "step": 1090, |
| "step_time": 8.175771025242284 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 846.7, |
| "completions/max_terminated_length": 846.7, |
| "completions/mean_length": 312.44921875, |
| "completions/mean_terminated_length": 312.44921875, |
| "completions/min_length": 172.5, |
| "completions/min_terminated_length": 172.5, |
| "entropy": 0.08294395194388926, |
| "epoch": 2.355460385438972, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.2060546875, |
| "learning_rate": 9.8901e-06, |
| "loss": -0.0055, |
| "num_tokens": 68633479.0, |
| "reward": 1.0539062738418579, |
| "reward_std": 0.181705242395401, |
| "rewards/reward_accuracy/mean": 0.95390625, |
| "rewards/reward_accuracy/std": 0.181705242395401, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8409517645835876, |
| "sampling/importance_sampling_ratio/mean": 0.9918436288833619, |
| "sampling/importance_sampling_ratio/min": 0.4342606008052826, |
| "sampling/sampling_logp_difference/max": 0.3439224541187286, |
| "sampling/sampling_logp_difference/mean": 0.0026392599334940313, |
| "step": 1100, |
| "step_time": 10.356933535449206 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 975.2, |
| "completions/max_terminated_length": 975.2, |
| "completions/mean_length": 331.753125, |
| "completions/mean_terminated_length": 331.753125, |
| "completions/min_length": 180.9, |
| "completions/min_terminated_length": 180.9, |
| "entropy": 0.07791998139582575, |
| "epoch": 2.3768736616702357, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.302734375, |
| "learning_rate": 9.8891e-06, |
| "loss": 0.0031, |
| "num_tokens": 69318091.0, |
| "reward": 1.0460937738418579, |
| "reward_std": 0.17950820550322533, |
| "rewards/reward_accuracy/mean": 0.94609375, |
| "rewards/reward_accuracy/std": 0.17950820550322533, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8677935600280762, |
| "sampling/importance_sampling_ratio/mean": 1.0076301991939545, |
| "sampling/importance_sampling_ratio/min": 0.422925665974617, |
| "sampling/sampling_logp_difference/max": 0.3654286086559296, |
| "sampling/sampling_logp_difference/mean": 0.002598200156353414, |
| "step": 1110, |
| "step_time": 11.660536265000701 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 733.6, |
| "completions/max_terminated_length": 733.6, |
| "completions/mean_length": 327.00546875, |
| "completions/mean_terminated_length": 327.00546875, |
| "completions/min_length": 182.5, |
| "completions/min_terminated_length": 182.5, |
| "entropy": 0.08168157017789782, |
| "epoch": 2.398286937901499, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.1962890625, |
| "learning_rate": 9.888100000000001e-06, |
| "loss": 0.0016, |
| "num_tokens": 69997482.0, |
| "reward": 1.048437523841858, |
| "reward_std": 0.19420834705233575, |
| "rewards/reward_accuracy/mean": 0.9484375, |
| "rewards/reward_accuracy/std": 0.19420834705233575, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8929945826530457, |
| "sampling/importance_sampling_ratio/mean": 0.9941560864448548, |
| "sampling/importance_sampling_ratio/min": 0.4298032999038696, |
| "sampling/sampling_logp_difference/max": 0.3947967767715454, |
| "sampling/sampling_logp_difference/mean": 0.0027558892499655483, |
| "step": 1120, |
| "step_time": 9.568299278710038 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 798.9, |
| "completions/max_terminated_length": 798.9, |
| "completions/mean_length": 353.28125, |
| "completions/mean_terminated_length": 353.28125, |
| "completions/min_length": 197.0, |
| "completions/min_terminated_length": 197.0, |
| "entropy": 0.08312311947811395, |
| "epoch": 2.4197002141327624, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.11083984375, |
| "learning_rate": 9.8871e-06, |
| "loss": 0.0031, |
| "num_tokens": 70711690.0, |
| "reward": 1.0217969000339509, |
| "reward_std": 0.24880455583333969, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.24859188944101335, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006225345656275749, |
| "sampling/importance_sampling_ratio/max": 2.0040427923202513, |
| "sampling/importance_sampling_ratio/mean": 0.9978883028030395, |
| "sampling/importance_sampling_ratio/min": 0.3736264020204544, |
| "sampling/sampling_logp_difference/max": 0.39291518926620483, |
| "sampling/sampling_logp_difference/mean": 0.0028895307099446655, |
| "step": 1130, |
| "step_time": 9.99124787794426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 834.7, |
| "completions/max_terminated_length": 834.7, |
| "completions/mean_length": 351.7078125, |
| "completions/mean_terminated_length": 351.7078125, |
| "completions/min_length": 183.9, |
| "completions/min_terminated_length": 183.9, |
| "entropy": 0.08060317595954984, |
| "epoch": 2.4411134903640255, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.220703125, |
| "learning_rate": 9.8861e-06, |
| "loss": 0.0031, |
| "num_tokens": 71420500.0, |
| "reward": 1.037500023841858, |
| "reward_std": 0.20517989844083787, |
| "rewards/reward_accuracy/mean": 0.9375, |
| "rewards/reward_accuracy/std": 0.20517989844083787, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.114181661605835, |
| "sampling/importance_sampling_ratio/mean": 1.005430907011032, |
| "sampling/importance_sampling_ratio/min": 0.4228169143199921, |
| "sampling/sampling_logp_difference/max": 0.3839725375175476, |
| "sampling/sampling_logp_difference/mean": 0.0027914856560528277, |
| "step": 1140, |
| "step_time": 10.13563480800949 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 714.6, |
| "completions/max_terminated_length": 714.6, |
| "completions/mean_length": 350.74453125, |
| "completions/mean_terminated_length": 350.74453125, |
| "completions/min_length": 194.9, |
| "completions/min_terminated_length": 194.9, |
| "entropy": 0.08318151326384396, |
| "epoch": 2.462526766595289, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.885100000000001e-06, |
| "loss": -0.0031, |
| "num_tokens": 72127469.0, |
| "reward": 1.0358593940734864, |
| "reward_std": 0.20318207144737244, |
| "rewards/reward_accuracy/mean": 0.9359375, |
| "rewards/reward_accuracy/std": 0.20284141153097152, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.8955372810363769, |
| "sampling/importance_sampling_ratio/mean": 0.9961326539516449, |
| "sampling/importance_sampling_ratio/min": 0.37985644638538363, |
| "sampling/sampling_logp_difference/max": 0.38309550285339355, |
| "sampling/sampling_logp_difference/mean": 0.0028706039767712353, |
| "step": 1150, |
| "step_time": 9.542928419727833 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 1026.7, |
| "completions/max_terminated_length": 936.7, |
| "completions/mean_length": 368.36796875, |
| "completions/mean_terminated_length": 366.256298828125, |
| "completions/min_length": 194.9, |
| "completions/min_terminated_length": 194.9, |
| "entropy": 0.08709894071798771, |
| "epoch": 2.4839400428265526, |
| "frac_reward_zero_std": 0.84375, |
| "grad_norm": 0.3359375, |
| "learning_rate": 9.8841e-06, |
| "loss": 0.0015, |
| "num_tokens": 72860228.0, |
| "reward": 1.0357812821865082, |
| "reward_std": 0.242416812479496, |
| "rewards/reward_accuracy/mean": 0.9359375, |
| "rewards/reward_accuracy/std": 0.24185388535261154, |
| "rewards/reward_format/mean": 0.09984375238418579, |
| "rewards/reward_format/std": 0.001767767034471035, |
| "sampling/importance_sampling_ratio/max": 1.8464398622512816, |
| "sampling/importance_sampling_ratio/mean": 0.9988512694835663, |
| "sampling/importance_sampling_ratio/min": 0.38091603517532346, |
| "sampling/sampling_logp_difference/max": 0.3428805410861969, |
| "sampling/sampling_logp_difference/mean": 0.002888034959323704, |
| "step": 1160, |
| "step_time": 12.129660980496556 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 938.3, |
| "completions/max_terminated_length": 938.3, |
| "completions/mean_length": 340.92578125, |
| "completions/mean_terminated_length": 340.92578125, |
| "completions/min_length": 184.2, |
| "completions/min_terminated_length": 184.2, |
| "entropy": 0.08761966610327362, |
| "epoch": 2.505353319057816, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.1328125, |
| "learning_rate": 9.8831e-06, |
| "loss": -0.0027, |
| "num_tokens": 73555101.0, |
| "reward": 1.0514844059944153, |
| "reward_std": 0.2071546010673046, |
| "rewards/reward_accuracy/mean": 0.9515625, |
| "rewards/reward_accuracy/std": 0.20685592517256737, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0008838835172355175, |
| "sampling/importance_sampling_ratio/max": 1.9456478476524353, |
| "sampling/importance_sampling_ratio/mean": 0.9951296627521515, |
| "sampling/importance_sampling_ratio/min": 0.4514074593782425, |
| "sampling/sampling_logp_difference/max": 0.3684475660324097, |
| "sampling/sampling_logp_difference/mean": 0.0028999225934967397, |
| "step": 1170, |
| "step_time": 11.21017906339839 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 672.0, |
| "completions/max_terminated_length": 672.0, |
| "completions/mean_length": 326.2421875, |
| "completions/mean_terminated_length": 326.2421875, |
| "completions/min_length": 181.5, |
| "completions/min_terminated_length": 181.5, |
| "entropy": 0.08221625997684896, |
| "epoch": 2.526766595289079, |
| "frac_reward_zero_std": 0.93125, |
| "grad_norm": 0.16015625, |
| "learning_rate": 9.882100000000001e-06, |
| "loss": 0.0029, |
| "num_tokens": 74230395.0, |
| "reward": 1.0664062738418578, |
| "reward_std": 0.16644030809402466, |
| "rewards/reward_accuracy/mean": 0.96640625, |
| "rewards/reward_accuracy/std": 0.16644030809402466, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.840121591091156, |
| "sampling/importance_sampling_ratio/mean": 1.0035568177700043, |
| "sampling/importance_sampling_ratio/min": 0.5274126648902893, |
| "sampling/sampling_logp_difference/max": 0.41821449995040894, |
| "sampling/sampling_logp_difference/mean": 0.0027541376184672117, |
| "step": 1180, |
| "step_time": 8.86348119592294 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 748.5, |
| "completions/max_terminated_length": 748.5, |
| "completions/mean_length": 346.31796875, |
| "completions/mean_terminated_length": 346.31796875, |
| "completions/min_length": 186.8, |
| "completions/min_terminated_length": 186.8, |
| "entropy": 0.08588558440096676, |
| "epoch": 2.5481798715203428, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.228515625, |
| "learning_rate": 9.881100000000001e-06, |
| "loss": 0.0003, |
| "num_tokens": 74935826.0, |
| "reward": 1.051523458957672, |
| "reward_std": 0.19308943077921867, |
| "rewards/reward_accuracy/mean": 0.9515625, |
| "rewards/reward_accuracy/std": 0.19299574121832846, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 2.0112762451171875, |
| "sampling/importance_sampling_ratio/mean": 1.0071306765079497, |
| "sampling/importance_sampling_ratio/min": 0.37426243275403975, |
| "sampling/sampling_logp_difference/max": 0.5453690588474274, |
| "sampling/sampling_logp_difference/mean": 0.002894169115461409, |
| "step": 1190, |
| "step_time": 9.430566076003014 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 809.0, |
| "completions/max_terminated_length": 809.0, |
| "completions/mean_length": 312.315625, |
| "completions/mean_terminated_length": 312.315625, |
| "completions/min_length": 178.0, |
| "completions/min_terminated_length": 178.0, |
| "entropy": 0.07839165988843888, |
| "epoch": 2.569593147751606, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.208984375, |
| "learning_rate": 9.880100000000002e-06, |
| "loss": 0.0004, |
| "num_tokens": 75594758.0, |
| "reward": 1.0655859589576722, |
| "reward_std": 0.17501178458333017, |
| "rewards/reward_accuracy/mean": 0.965625, |
| "rewards/reward_accuracy/std": 0.1748311184346676, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.8693220019340515, |
| "sampling/importance_sampling_ratio/mean": 0.9993482410907746, |
| "sampling/importance_sampling_ratio/min": 0.47223553359508513, |
| "sampling/sampling_logp_difference/max": 0.38192314803600313, |
| "sampling/sampling_logp_difference/mean": 0.002720042713917792, |
| "step": 1200, |
| "step_time": 10.062140053324402 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 624.7, |
| "completions/max_terminated_length": 624.7, |
| "completions/mean_length": 314.62109375, |
| "completions/mean_terminated_length": 314.62109375, |
| "completions/min_length": 178.0, |
| "completions/min_terminated_length": 178.0, |
| "entropy": 0.07802212219685316, |
| "epoch": 2.5910064239828694, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.126953125, |
| "learning_rate": 9.8791e-06, |
| "loss": 0.0021, |
| "num_tokens": 76256489.0, |
| "reward": 1.0421484649181365, |
| "reward_std": 0.22055592089891435, |
| "rewards/reward_accuracy/mean": 0.9421875, |
| "rewards/reward_accuracy/std": 0.22044544965028762, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 2.041950690746307, |
| "sampling/importance_sampling_ratio/mean": 0.9945596098899842, |
| "sampling/importance_sampling_ratio/min": 0.4890503704547882, |
| "sampling/sampling_logp_difference/max": 0.4137039422988892, |
| "sampling/sampling_logp_difference/mean": 0.0027024110546335577, |
| "step": 1210, |
| "step_time": 8.454517868952825 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 596.4, |
| "completions/max_terminated_length": 596.4, |
| "completions/mean_length": 317.846875, |
| "completions/mean_terminated_length": 317.846875, |
| "completions/min_length": 178.2, |
| "completions/min_terminated_length": 178.2, |
| "entropy": 0.07332060048356652, |
| "epoch": 2.612419700214133, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.103515625, |
| "learning_rate": 9.878100000000001e-06, |
| "loss": -0.0003, |
| "num_tokens": 76922229.0, |
| "reward": 1.0585937738418578, |
| "reward_std": 0.16497317627072333, |
| "rewards/reward_accuracy/mean": 0.95859375, |
| "rewards/reward_accuracy/std": 0.16497317627072333, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.884109663963318, |
| "sampling/importance_sampling_ratio/mean": 0.9925716817378998, |
| "sampling/importance_sampling_ratio/min": 0.4664496570825577, |
| "sampling/sampling_logp_difference/max": 0.37606627941131593, |
| "sampling/sampling_logp_difference/mean": 0.0025874980725347995, |
| "step": 1220, |
| "step_time": 7.993150155153126 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 566.5, |
| "completions/max_terminated_length": 566.5, |
| "completions/mean_length": 304.36640625, |
| "completions/mean_terminated_length": 304.36640625, |
| "completions/min_length": 166.7, |
| "completions/min_terminated_length": 166.7, |
| "entropy": 0.07706191691104322, |
| "epoch": 2.633832976445396, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.162109375, |
| "learning_rate": 9.8771e-06, |
| "loss": 0.001, |
| "num_tokens": 77569658.0, |
| "reward": 1.053125023841858, |
| "reward_std": 0.1919751785695553, |
| "rewards/reward_accuracy/mean": 0.953125, |
| "rewards/reward_accuracy/std": 0.1919751785695553, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9988874316215515, |
| "sampling/importance_sampling_ratio/mean": 1.0073664367198945, |
| "sampling/importance_sampling_ratio/min": 0.5318764299154282, |
| "sampling/sampling_logp_difference/max": 0.3568783521652222, |
| "sampling/sampling_logp_difference/mean": 0.002705926960334182, |
| "step": 1230, |
| "step_time": 8.069069889793173 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 963.5, |
| "completions/max_terminated_length": 963.5, |
| "completions/mean_length": 341.73828125, |
| "completions/mean_terminated_length": 341.73828125, |
| "completions/min_length": 184.0, |
| "completions/min_terminated_length": 184.0, |
| "entropy": 0.08499204977415502, |
| "epoch": 2.6552462526766596, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.251953125, |
| "learning_rate": 9.8761e-06, |
| "loss": -0.0021, |
| "num_tokens": 78266379.0, |
| "reward": 1.0390625238418578, |
| "reward_std": 0.22909086495637893, |
| "rewards/reward_accuracy/mean": 0.9390625, |
| "rewards/reward_accuracy/std": 0.22909086495637893, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.967051374912262, |
| "sampling/importance_sampling_ratio/mean": 1.001557755470276, |
| "sampling/importance_sampling_ratio/min": 0.3729872371796915, |
| "sampling/sampling_logp_difference/max": 2.0925530433654784, |
| "sampling/sampling_logp_difference/mean": 0.002837384701706469, |
| "step": 1240, |
| "step_time": 11.349837810546159 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 870.6, |
| "completions/max_terminated_length": 870.6, |
| "completions/mean_length": 345.73984375, |
| "completions/mean_terminated_length": 345.73984375, |
| "completions/min_length": 200.0, |
| "completions/min_terminated_length": 200.0, |
| "entropy": 0.08017215605359525, |
| "epoch": 2.6766595289079227, |
| "frac_reward_zero_std": 0.9, |
| "grad_norm": 0.287109375, |
| "learning_rate": 9.875100000000001e-06, |
| "loss": 0.0006, |
| "num_tokens": 78969678.0, |
| "reward": 1.055468773841858, |
| "reward_std": 0.1820686012506485, |
| "rewards/reward_accuracy/mean": 0.95546875, |
| "rewards/reward_accuracy/std": 0.1820686012506485, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9609744548797607, |
| "sampling/importance_sampling_ratio/mean": 1.0008956015110015, |
| "sampling/importance_sampling_ratio/min": 0.40326684442115945, |
| "sampling/sampling_logp_difference/max": 0.9730170011520386, |
| "sampling/sampling_logp_difference/mean": 0.002679547434672713, |
| "step": 1250, |
| "step_time": 10.552440194785595 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 713.4, |
| "completions/max_terminated_length": 713.4, |
| "completions/mean_length": 342.3375, |
| "completions/mean_terminated_length": 342.3375, |
| "completions/min_length": 203.6, |
| "completions/min_terminated_length": 203.6, |
| "entropy": 0.08833467878866941, |
| "epoch": 2.6980728051391862, |
| "frac_reward_zero_std": 0.90625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.874100000000001e-06, |
| "loss": -0.0008, |
| "num_tokens": 79668854.0, |
| "reward": 1.0530859589576722, |
| "reward_std": 0.1750552595127374, |
| "rewards/reward_accuracy/mean": 0.953125, |
| "rewards/reward_accuracy/std": 0.1746133178472519, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.8788799047470093, |
| "sampling/importance_sampling_ratio/mean": 0.9963806211948395, |
| "sampling/importance_sampling_ratio/min": 0.4560327798128128, |
| "sampling/sampling_logp_difference/max": 0.36841505765914917, |
| "sampling/sampling_logp_difference/mean": 0.0030034594237804413, |
| "step": 1260, |
| "step_time": 9.35989262117073 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 685.3, |
| "completions/max_terminated_length": 685.3, |
| "completions/mean_length": 354.2640625, |
| "completions/mean_terminated_length": 354.2640625, |
| "completions/min_length": 201.0, |
| "completions/min_terminated_length": 201.0, |
| "entropy": 0.08774566981010139, |
| "epoch": 2.71948608137045, |
| "frac_reward_zero_std": 0.9, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8731e-06, |
| "loss": -0.0015, |
| "num_tokens": 80381328.0, |
| "reward": 1.0546875238418578, |
| "reward_std": 0.16447920128703117, |
| "rewards/reward_accuracy/mean": 0.9546875, |
| "rewards/reward_accuracy/std": 0.16447920128703117, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0336657524108888, |
| "sampling/importance_sampling_ratio/mean": 1.011208724975586, |
| "sampling/importance_sampling_ratio/min": 0.4497158318758011, |
| "sampling/sampling_logp_difference/max": 0.3409749746322632, |
| "sampling/sampling_logp_difference/mean": 0.002926108567044139, |
| "step": 1270, |
| "step_time": 8.619230843987316 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 958.6, |
| "completions/max_terminated_length": 958.6, |
| "completions/mean_length": 358.69921875, |
| "completions/mean_terminated_length": 358.69921875, |
| "completions/min_length": 210.6, |
| "completions/min_terminated_length": 210.6, |
| "entropy": 0.08935087500140071, |
| "epoch": 2.7408993576017133, |
| "frac_reward_zero_std": 0.86875, |
| "grad_norm": 0.15234375, |
| "learning_rate": 9.872100000000002e-06, |
| "loss": -0.0021, |
| "num_tokens": 81101991.0, |
| "reward": 1.0507812738418578, |
| "reward_std": 0.2018032416701317, |
| "rewards/reward_accuracy/mean": 0.95078125, |
| "rewards/reward_accuracy/std": 0.2018032416701317, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9050111055374146, |
| "sampling/importance_sampling_ratio/mean": 1.0004482209682464, |
| "sampling/importance_sampling_ratio/min": 0.484862744808197, |
| "sampling/sampling_logp_difference/max": 0.3985797524452209, |
| "sampling/sampling_logp_difference/mean": 0.0030397461261600254, |
| "step": 1280, |
| "step_time": 11.516265662154183 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 631.2, |
| "completions/max_terminated_length": 631.2, |
| "completions/mean_length": 350.25, |
| "completions/mean_terminated_length": 350.25, |
| "completions/min_length": 202.3, |
| "completions/min_terminated_length": 202.3, |
| "entropy": 0.07935340562835336, |
| "epoch": 2.7623126338329764, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.189453125, |
| "learning_rate": 9.871100000000001e-06, |
| "loss": -0.0043, |
| "num_tokens": 81812743.0, |
| "reward": 1.068750023841858, |
| "reward_std": 0.15125490203499795, |
| "rewards/reward_accuracy/mean": 0.96875, |
| "rewards/reward_accuracy/std": 0.15125490203499795, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0169615387916564, |
| "sampling/importance_sampling_ratio/mean": 0.9952198743820191, |
| "sampling/importance_sampling_ratio/min": 0.46572147905826566, |
| "sampling/sampling_logp_difference/max": 0.4421790957450867, |
| "sampling/sampling_logp_difference/mean": 0.0029164568288251756, |
| "step": 1290, |
| "step_time": 8.581981379771605 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 863.0, |
| "completions/max_terminated_length": 863.0, |
| "completions/mean_length": 344.334375, |
| "completions/mean_terminated_length": 344.334375, |
| "completions/min_length": 197.1, |
| "completions/min_terminated_length": 197.1, |
| "entropy": 0.08210056389216333, |
| "epoch": 2.78372591006424, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8701e-06, |
| "loss": -0.0013, |
| "num_tokens": 82514227.0, |
| "reward": 1.040507835149765, |
| "reward_std": 0.2189115732908249, |
| "rewards/reward_accuracy/mean": 0.940625, |
| "rewards/reward_accuracy/std": 0.2186038762331009, |
| "rewards/reward_format/mean": 0.09988281354308129, |
| "rewards/reward_format/std": 0.0007594143506139516, |
| "sampling/importance_sampling_ratio/max": 2.0759007692337037, |
| "sampling/importance_sampling_ratio/mean": 1.0040241956710816, |
| "sampling/importance_sampling_ratio/min": 0.38231505155563356, |
| "sampling/sampling_logp_difference/max": 0.38071630895137787, |
| "sampling/sampling_logp_difference/mean": 0.003062532260082662, |
| "step": 1300, |
| "step_time": 10.30869936766103 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 706.7, |
| "completions/max_terminated_length": 706.7, |
| "completions/mean_length": 326.325, |
| "completions/mean_terminated_length": 326.325, |
| "completions/min_length": 183.5, |
| "completions/min_terminated_length": 183.5, |
| "entropy": 0.07457869255449623, |
| "epoch": 2.805139186295503, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8691e-06, |
| "loss": 0.0039, |
| "num_tokens": 83192283.0, |
| "reward": 1.080468773841858, |
| "reward_std": 0.11533628031611443, |
| "rewards/reward_accuracy/mean": 0.98046875, |
| "rewards/reward_accuracy/std": 0.11533628031611443, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8987989187240601, |
| "sampling/importance_sampling_ratio/mean": 0.9928642272949219, |
| "sampling/importance_sampling_ratio/min": 0.46463883817195895, |
| "sampling/sampling_logp_difference/max": 0.32871644496917723, |
| "sampling/sampling_logp_difference/mean": 0.002699966193176806, |
| "step": 1310, |
| "step_time": 9.153831074992194 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1031.0, |
| "completions/max_terminated_length": 1031.0, |
| "completions/mean_length": 325.6703125, |
| "completions/mean_terminated_length": 325.6703125, |
| "completions/min_length": 178.6, |
| "completions/min_terminated_length": 178.6, |
| "entropy": 0.07806437211111188, |
| "epoch": 2.8265524625267666, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.1025390625, |
| "learning_rate": 9.868100000000001e-06, |
| "loss": -0.0003, |
| "num_tokens": 83866973.0, |
| "reward": 1.0491797089576722, |
| "reward_std": 0.19839874282479286, |
| "rewards/reward_accuracy/mean": 0.94921875, |
| "rewards/reward_accuracy/std": 0.19840871170163155, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.9467016339302063, |
| "sampling/importance_sampling_ratio/mean": 0.9990678548812866, |
| "sampling/importance_sampling_ratio/min": 0.42225509136915207, |
| "sampling/sampling_logp_difference/max": 0.5470814347267151, |
| "sampling/sampling_logp_difference/mean": 0.002948223613202572, |
| "step": 1320, |
| "step_time": 12.024973524315282 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 858.6, |
| "completions/max_terminated_length": 858.6, |
| "completions/mean_length": 293.9265625, |
| "completions/mean_terminated_length": 293.9265625, |
| "completions/min_length": 164.1, |
| "completions/min_terminated_length": 164.1, |
| "entropy": 0.07185970705468207, |
| "epoch": 2.84796573875803, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.11865234375, |
| "learning_rate": 9.8671e-06, |
| "loss": 0.0018, |
| "num_tokens": 84499783.0, |
| "reward": 1.048398458957672, |
| "reward_std": 0.2032108038663864, |
| "rewards/reward_accuracy/mean": 0.9484375, |
| "rewards/reward_accuracy/std": 0.20321595817804336, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 2.1334774494171143, |
| "sampling/importance_sampling_ratio/mean": 1.0080353975296021, |
| "sampling/importance_sampling_ratio/min": 0.44090257585048676, |
| "sampling/sampling_logp_difference/max": 0.5914352059364318, |
| "sampling/sampling_logp_difference/mean": 0.002739799418486655, |
| "step": 1330, |
| "step_time": 10.607592740794644 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 599.9, |
| "completions/max_terminated_length": 599.9, |
| "completions/mean_length": 290.16796875, |
| "completions/mean_terminated_length": 290.16796875, |
| "completions/min_length": 148.7, |
| "completions/min_terminated_length": 148.7, |
| "entropy": 0.07274656272493303, |
| "epoch": 2.8693790149892933, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.24609375, |
| "learning_rate": 9.8661e-06, |
| "loss": -0.0044, |
| "num_tokens": 85128574.0, |
| "reward": 1.0421875238418579, |
| "reward_std": 0.2102555438876152, |
| "rewards/reward_accuracy/mean": 0.9421875, |
| "rewards/reward_accuracy/std": 0.2102555438876152, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8882685542106628, |
| "sampling/importance_sampling_ratio/mean": 1.008824783563614, |
| "sampling/importance_sampling_ratio/min": 0.44516457840800283, |
| "sampling/sampling_logp_difference/max": 0.4197180211544037, |
| "sampling/sampling_logp_difference/mean": 0.0028648935724049805, |
| "step": 1340, |
| "step_time": 8.176954083051532 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00078125, |
| "completions/max_length": 876.9, |
| "completions/max_terminated_length": 863.7, |
| "completions/mean_length": 300.46015625, |
| "completions/mean_terminated_length": 298.33921508789064, |
| "completions/min_length": 164.4, |
| "completions/min_terminated_length": 164.4, |
| "entropy": 0.072030695900321, |
| "epoch": 2.890792291220557, |
| "frac_reward_zero_std": 0.81875, |
| "grad_norm": 0.419921875, |
| "learning_rate": 9.865100000000001e-06, |
| "loss": 0.0006, |
| "num_tokens": 85772987.0, |
| "reward": 1.0381250262260437, |
| "reward_std": 0.23116080984473228, |
| "rewards/reward_accuracy/mean": 0.93828125, |
| "rewards/reward_accuracy/std": 0.23067262694239615, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012450690381228923, |
| "sampling/importance_sampling_ratio/max": 2.022037720680237, |
| "sampling/importance_sampling_ratio/mean": 0.9964764177799225, |
| "sampling/importance_sampling_ratio/min": 0.35306877121329305, |
| "sampling/sampling_logp_difference/max": 0.7961321771144867, |
| "sampling/sampling_logp_difference/mean": 0.0028138081775978207, |
| "step": 1350, |
| "step_time": 11.209532467182726 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 689.1, |
| "completions/max_terminated_length": 689.1, |
| "completions/mean_length": 305.17109375, |
| "completions/mean_terminated_length": 305.17109375, |
| "completions/min_length": 164.7, |
| "completions/min_terminated_length": 164.7, |
| "entropy": 0.06921829774510116, |
| "epoch": 2.91220556745182, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.30078125, |
| "learning_rate": 9.864100000000001e-06, |
| "loss": -0.0039, |
| "num_tokens": 86425870.0, |
| "reward": 1.039804708957672, |
| "reward_std": 0.21331144794821738, |
| "rewards/reward_accuracy/mean": 0.93984375, |
| "rewards/reward_accuracy/std": 0.21332000717520713, |
| "rewards/reward_format/mean": 0.0999609388411045, |
| "rewards/reward_format/std": 0.00044194171205163, |
| "sampling/importance_sampling_ratio/max": 1.9660413980484008, |
| "sampling/importance_sampling_ratio/mean": 1.0018531382083893, |
| "sampling/importance_sampling_ratio/min": 0.419755095243454, |
| "sampling/sampling_logp_difference/max": 0.3945651054382324, |
| "sampling/sampling_logp_difference/mean": 0.0026544601656496524, |
| "step": 1360, |
| "step_time": 8.99832956190221 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 554.7, |
| "completions/max_terminated_length": 554.7, |
| "completions/mean_length": 290.271875, |
| "completions/mean_terminated_length": 290.271875, |
| "completions/min_length": 160.2, |
| "completions/min_terminated_length": 160.2, |
| "entropy": 0.06351722655817867, |
| "epoch": 2.9336188436830835, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8631e-06, |
| "loss": 0.0019, |
| "num_tokens": 87054402.0, |
| "reward": 1.0734375238418579, |
| "reward_std": 0.14015127643942832, |
| "rewards/reward_accuracy/mean": 0.9734375, |
| "rewards/reward_accuracy/std": 0.14015127643942832, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8296015739440918, |
| "sampling/importance_sampling_ratio/mean": 1.005638563632965, |
| "sampling/importance_sampling_ratio/min": 0.488324898481369, |
| "sampling/sampling_logp_difference/max": 0.36103426218032836, |
| "sampling/sampling_logp_difference/mean": 0.002417617873288691, |
| "step": 1370, |
| "step_time": 7.836227551661432 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 571.2, |
| "completions/max_terminated_length": 571.2, |
| "completions/mean_length": 285.44765625, |
| "completions/mean_terminated_length": 285.44765625, |
| "completions/min_length": 158.5, |
| "completions/min_terminated_length": 158.5, |
| "entropy": 0.0635203366400674, |
| "epoch": 2.955032119914347, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.181640625, |
| "learning_rate": 9.862100000000002e-06, |
| "loss": -0.0006, |
| "num_tokens": 87677679.0, |
| "reward": 1.0421875238418579, |
| "reward_std": 0.22069814950227737, |
| "rewards/reward_accuracy/mean": 0.9421875, |
| "rewards/reward_accuracy/std": 0.22069814950227737, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8875408053398133, |
| "sampling/importance_sampling_ratio/mean": 0.9922976195812225, |
| "sampling/importance_sampling_ratio/min": 0.5207251012325287, |
| "sampling/sampling_logp_difference/max": 0.3793253242969513, |
| "sampling/sampling_logp_difference/mean": 0.0023526079021394253, |
| "step": 1380, |
| "step_time": 8.122663669334724 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 566.8, |
| "completions/max_terminated_length": 566.8, |
| "completions/mean_length": 285.8625, |
| "completions/mean_terminated_length": 285.8625, |
| "completions/min_length": 156.8, |
| "completions/min_terminated_length": 156.8, |
| "entropy": 0.06256461564917118, |
| "epoch": 2.9764453961456105, |
| "frac_reward_zero_std": 0.88125, |
| "grad_norm": 0.0, |
| "learning_rate": 9.861100000000001e-06, |
| "loss": -0.0009, |
| "num_tokens": 88299791.0, |
| "reward": 1.068750023841858, |
| "reward_std": 0.15067666098475457, |
| "rewards/reward_accuracy/mean": 0.96875, |
| "rewards/reward_accuracy/std": 0.15067666098475457, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.903975248336792, |
| "sampling/importance_sampling_ratio/mean": 1.002293837070465, |
| "sampling/importance_sampling_ratio/min": 0.47189922630786896, |
| "sampling/sampling_logp_difference/max": 0.3698516845703125, |
| "sampling/sampling_logp_difference/mean": 0.0023362166830338538, |
| "step": 1390, |
| "step_time": 7.800217655627057 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 692.2, |
| "completions/max_terminated_length": 692.2, |
| "completions/mean_length": 296.07578125, |
| "completions/mean_terminated_length": 296.07578125, |
| "completions/min_length": 163.6, |
| "completions/min_terminated_length": 163.6, |
| "entropy": 0.06756354942917824, |
| "epoch": 2.9978586723768736, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.20703125, |
| "learning_rate": 9.8601e-06, |
| "loss": -0.0078, |
| "num_tokens": 88936744.0, |
| "reward": 1.0507812738418578, |
| "reward_std": 0.19577359780669212, |
| "rewards/reward_accuracy/mean": 0.95078125, |
| "rewards/reward_accuracy/std": 0.19577359780669212, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8318641901016235, |
| "sampling/importance_sampling_ratio/mean": 0.9952113807201386, |
| "sampling/importance_sampling_ratio/min": 0.38276190012693406, |
| "sampling/sampling_logp_difference/max": 0.3824802815914154, |
| "sampling/sampling_logp_difference/mean": 0.0025198230519890784, |
| "step": 1400, |
| "step_time": 9.158693155134097 |
| }, |
| { |
| "clip_ratio/high_max": 1.4568764891009778e-05, |
| "clip_ratio/high_mean": 3.6421912227524445e-06, |
| "clip_ratio/low_mean": 3.996163650299423e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 7.638354873051867e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 525.2, |
| "completions/max_terminated_length": 525.2, |
| "completions/mean_length": 313.49375, |
| "completions/mean_terminated_length": 313.49375, |
| "completions/min_length": 172.6, |
| "completions/min_terminated_length": 172.6, |
| "entropy": 0.07267224765382707, |
| "epoch": 1.5096359743040684, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.55078125, |
| "learning_rate": 9.8591e-06, |
| "loss": 0.0021, |
| "num_tokens": 89273796.0, |
| "reward": 1.059375023841858, |
| "reward_std": 0.15481940805912017, |
| "rewards/reward_accuracy/mean": 0.959375, |
| "rewards/reward_accuracy/std": 0.15481941252946854, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7013391733169556, |
| "sampling/importance_sampling_ratio/mean": 1.0019705176353455, |
| "sampling/importance_sampling_ratio/min": 0.5237392753362655, |
| "sampling/sampling_logp_difference/max": 0.3254459023475647, |
| "sampling/sampling_logp_difference/mean": 0.002789818309247494, |
| "step": 1410, |
| "step_time": 7.13007674338296 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 454.3, |
| "completions/max_terminated_length": 454.3, |
| "completions/mean_length": 283.9453125, |
| "completions/mean_terminated_length": 283.9453125, |
| "completions/min_length": 162.4, |
| "completions/min_terminated_length": 162.4, |
| "entropy": 0.06823760028928519, |
| "epoch": 1.5203426124197001, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.474609375, |
| "learning_rate": 9.8581e-06, |
| "loss": 0.0037, |
| "num_tokens": 89585537.0, |
| "reward": 1.0781250238418578, |
| "reward_std": 0.100810107588768, |
| "rewards/reward_accuracy/mean": 0.978125, |
| "rewards/reward_accuracy/std": 0.10081011354923249, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.721087145805359, |
| "sampling/importance_sampling_ratio/mean": 0.9957422256469727, |
| "sampling/importance_sampling_ratio/min": 0.5624752104282379, |
| "sampling/sampling_logp_difference/max": 0.36021349430084226, |
| "sampling/sampling_logp_difference/mean": 0.002582652703858912, |
| "step": 1420, |
| "step_time": 6.363762807892636 |
| }, |
| { |
| "clip_ratio/high_max": 1.6025641525629907e-05, |
| "clip_ratio/high_mean": 4.006410381407477e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.006410381407477e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 470.3, |
| "completions/max_terminated_length": 470.3, |
| "completions/mean_length": 274.1578125, |
| "completions/mean_terminated_length": 274.1578125, |
| "completions/min_length": 168.9, |
| "completions/min_terminated_length": 168.9, |
| "entropy": 0.05655218656174839, |
| "epoch": 1.531049250535332, |
| "frac_reward_zero_std": 0.975, |
| "grad_norm": 0.0, |
| "learning_rate": 9.857100000000001e-06, |
| "loss": -0.0001, |
| "num_tokens": 89887174.0, |
| "reward": 1.071875023841858, |
| "reward_std": 0.09166666865348816, |
| "rewards/reward_accuracy/mean": 0.971875, |
| "rewards/reward_accuracy/std": 0.09166666865348816, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.576025915145874, |
| "sampling/importance_sampling_ratio/mean": 1.012771213054657, |
| "sampling/importance_sampling_ratio/min": 0.6149522244930268, |
| "sampling/sampling_logp_difference/max": 0.288088059425354, |
| "sampling/sampling_logp_difference/mean": 0.0020975461695343254, |
| "step": 1430, |
| "step_time": 6.489046306908131 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 3.559225660865195e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.559225660865195e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 785.5, |
| "completions/max_terminated_length": 785.5, |
| "completions/mean_length": 306.8671875, |
| "completions/mean_terminated_length": 306.8671875, |
| "completions/min_length": 186.3, |
| "completions/min_terminated_length": 186.3, |
| "entropy": 0.07142699826508761, |
| "epoch": 1.5417558886509637, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.32421875, |
| "learning_rate": 9.8561e-06, |
| "loss": 0.0054, |
| "num_tokens": 90211553.0, |
| "reward": 1.048437523841858, |
| "reward_std": 0.18271250426769256, |
| "rewards/reward_accuracy/mean": 0.9484375, |
| "rewards/reward_accuracy/std": 0.18271250873804093, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.812452220916748, |
| "sampling/importance_sampling_ratio/mean": 1.0099130630493165, |
| "sampling/importance_sampling_ratio/min": 0.5242701947689057, |
| "sampling/sampling_logp_difference/max": 0.35053013563156127, |
| "sampling/sampling_logp_difference/mean": 0.002624026872217655, |
| "step": 1440, |
| "step_time": 9.329236005758867 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.2913223145005758e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.2913223145005758e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 713.7, |
| "completions/max_terminated_length": 713.7, |
| "completions/mean_length": 315.834375, |
| "completions/mean_terminated_length": 315.834375, |
| "completions/min_length": 196.7, |
| "completions/min_terminated_length": 196.7, |
| "entropy": 0.06896550043020397, |
| "epoch": 1.5524625267665952, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.41015625, |
| "learning_rate": 9.855100000000002e-06, |
| "loss": 0.0097, |
| "num_tokens": 90545447.0, |
| "reward": 1.0625000238418578, |
| "reward_std": 0.14929490983486177, |
| "rewards/reward_accuracy/mean": 0.9625, |
| "rewards/reward_accuracy/std": 0.14929491728544236, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7749225974082947, |
| "sampling/importance_sampling_ratio/mean": 1.0034668028354645, |
| "sampling/importance_sampling_ratio/min": 0.5281599760055542, |
| "sampling/sampling_logp_difference/max": 0.3764278531074524, |
| "sampling/sampling_logp_difference/mean": 0.002690198179334402, |
| "step": 1450, |
| "step_time": 8.682765920367093 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 710.9, |
| "completions/max_terminated_length": 710.9, |
| "completions/mean_length": 313.2984375, |
| "completions/mean_terminated_length": 313.2984375, |
| "completions/min_length": 177.8, |
| "completions/min_terminated_length": 177.8, |
| "entropy": 0.07382911436725408, |
| "epoch": 1.563169164882227, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.298828125, |
| "learning_rate": 9.854100000000001e-06, |
| "loss": 0.0, |
| "num_tokens": 90876454.0, |
| "reward": 1.048125034570694, |
| "reward_std": 0.17974907997995615, |
| "rewards/reward_accuracy/mean": 0.9484375, |
| "rewards/reward_accuracy/std": 0.17896783351898193, |
| "rewards/reward_format/mean": 0.0996875025331974, |
| "rewards/reward_format/std": 0.0021268405951559545, |
| "sampling/importance_sampling_ratio/max": 1.7648928046226502, |
| "sampling/importance_sampling_ratio/mean": 0.9862731218338012, |
| "sampling/importance_sampling_ratio/min": 0.5034876316785812, |
| "sampling/sampling_logp_difference/max": 0.36416717767715456, |
| "sampling/sampling_logp_difference/mean": 0.002803818532265723, |
| "step": 1460, |
| "step_time": 8.533526411326601 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.5687750419601797e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.5687750419601797e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 553.7, |
| "completions/max_terminated_length": 553.7, |
| "completions/mean_length": 281.54375, |
| "completions/mean_terminated_length": 281.54375, |
| "completions/min_length": 169.2, |
| "completions/min_terminated_length": 169.2, |
| "entropy": 0.07004948623944074, |
| "epoch": 1.5738758029978586, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.51171875, |
| "learning_rate": 9.8531e-06, |
| "loss": 0.0013, |
| "num_tokens": 91182482.0, |
| "reward": 1.053125023841858, |
| "reward_std": 0.17350118160247802, |
| "rewards/reward_accuracy/mean": 0.953125, |
| "rewards/reward_accuracy/std": 0.1735011890530586, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8077102303504944, |
| "sampling/importance_sampling_ratio/mean": 0.9815287172794342, |
| "sampling/importance_sampling_ratio/min": 0.462382498383522, |
| "sampling/sampling_logp_difference/max": 0.3806022882461548, |
| "sampling/sampling_logp_difference/mean": 0.002838394674472511, |
| "step": 1470, |
| "step_time": 7.272790736937895 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 486.7, |
| "completions/max_terminated_length": 486.7, |
| "completions/mean_length": 275.0390625, |
| "completions/mean_terminated_length": 275.0390625, |
| "completions/min_length": 163.6, |
| "completions/min_terminated_length": 163.6, |
| "entropy": 0.05895230381283909, |
| "epoch": 1.5845824411134903, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.34375, |
| "learning_rate": 9.852100000000002e-06, |
| "loss": 0.0049, |
| "num_tokens": 91486555.0, |
| "reward": 1.0561719059944152, |
| "reward_std": 0.14815565198659897, |
| "rewards/reward_accuracy/mean": 0.95625, |
| "rewards/reward_accuracy/std": 0.14814995378255844, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006250000558793544, |
| "sampling/importance_sampling_ratio/max": 1.8096882462501527, |
| "sampling/importance_sampling_ratio/mean": 1.0024614870548247, |
| "sampling/importance_sampling_ratio/min": 0.5027518600225449, |
| "sampling/sampling_logp_difference/max": 0.4047889709472656, |
| "sampling/sampling_logp_difference/mean": 0.002716872771270573, |
| "step": 1480, |
| "step_time": 6.636990210646763 |
| }, |
| { |
| "clip_ratio/high_max": 3.1172262970358135e-05, |
| "clip_ratio/high_mean": 7.793065742589534e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 7.793065742589534e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 515.5, |
| "completions/max_terminated_length": 515.5, |
| "completions/mean_length": 318.428125, |
| "completions/mean_terminated_length": 318.428125, |
| "completions/min_length": 177.8, |
| "completions/min_terminated_length": 177.8, |
| "entropy": 0.06482373329345137, |
| "epoch": 1.595289079229122, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.322265625, |
| "learning_rate": 9.851100000000001e-06, |
| "loss": -0.0029, |
| "num_tokens": 91822413.0, |
| "reward": 0.9951562762260437, |
| "reward_std": 0.2487994223833084, |
| "rewards/reward_accuracy/mean": 0.8953125, |
| "rewards/reward_accuracy/std": 0.2484452858567238, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0008768404833972455, |
| "sampling/importance_sampling_ratio/max": 2.125384974479675, |
| "sampling/importance_sampling_ratio/mean": 1.0124676764011382, |
| "sampling/importance_sampling_ratio/min": 0.4977218836545944, |
| "sampling/sampling_logp_difference/max": 0.4521936535835266, |
| "sampling/sampling_logp_difference/mean": 0.002916663698852062, |
| "step": 1490, |
| "step_time": 7.053972885059193 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 549.6, |
| "completions/max_terminated_length": 549.6, |
| "completions/mean_length": 316.253125, |
| "completions/mean_terminated_length": 316.253125, |
| "completions/min_length": 190.8, |
| "completions/min_terminated_length": 190.8, |
| "entropy": 0.06286788114812225, |
| "epoch": 1.6059957173447539, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.333984375, |
| "learning_rate": 9.8501e-06, |
| "loss": 0.006, |
| "num_tokens": 92151023.0, |
| "reward": 1.079687523841858, |
| "reward_std": 0.10383181273937225, |
| "rewards/reward_accuracy/mean": 0.9796875, |
| "rewards/reward_accuracy/std": 0.1038318157196045, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8306710362434386, |
| "sampling/importance_sampling_ratio/mean": 1.0021680474281311, |
| "sampling/importance_sampling_ratio/min": 0.5318920880556106, |
| "sampling/sampling_logp_difference/max": 0.3705809712409973, |
| "sampling/sampling_logp_difference/mean": 0.002577482839114964, |
| "step": 1500, |
| "step_time": 7.166918724495917 |
| }, |
| { |
| "clip_ratio/high_max": 4.1977800719905645e-05, |
| "clip_ratio/high_mean": 1.0494450179976411e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.0494450179976411e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 608.3, |
| "completions/max_terminated_length": 608.3, |
| "completions/mean_length": 334.515625, |
| "completions/mean_terminated_length": 334.515625, |
| "completions/min_length": 209.4, |
| "completions/min_terminated_length": 209.4, |
| "entropy": 0.06448490128386766, |
| "epoch": 1.6167023554603854, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.1748046875, |
| "learning_rate": 9.8491e-06, |
| "loss": -0.0008, |
| "num_tokens": 92497081.0, |
| "reward": 1.060937523841858, |
| "reward_std": 0.14996639788150787, |
| "rewards/reward_accuracy/mean": 0.9609375, |
| "rewards/reward_accuracy/std": 0.14996639788150787, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9107519149780274, |
| "sampling/importance_sampling_ratio/mean": 0.9928556859493256, |
| "sampling/importance_sampling_ratio/min": 0.38584725247928875, |
| "sampling/sampling_logp_difference/max": 1.3097064793109894, |
| "sampling/sampling_logp_difference/mean": 0.003018203633837402, |
| "step": 1510, |
| "step_time": 7.796595245087519 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.7265192582271993e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.7265192582271993e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 629.2, |
| "completions/max_terminated_length": 629.2, |
| "completions/mean_length": 332.05625, |
| "completions/mean_terminated_length": 332.05625, |
| "completions/min_length": 209.0, |
| "completions/min_terminated_length": 209.0, |
| "entropy": 0.07088577400427312, |
| "epoch": 1.627408993576017, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.275390625, |
| "learning_rate": 9.8481e-06, |
| "loss": 0.0024, |
| "num_tokens": 92837677.0, |
| "reward": 1.0234375238418578, |
| "reward_std": 0.2409310221672058, |
| "rewards/reward_accuracy/mean": 0.9234375, |
| "rewards/reward_accuracy/std": 0.24093102663755417, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0009332299232483, |
| "sampling/importance_sampling_ratio/mean": 1.0049149572849274, |
| "sampling/importance_sampling_ratio/min": 0.4818274974822998, |
| "sampling/sampling_logp_difference/max": 0.3970681428909302, |
| "sampling/sampling_logp_difference/mean": 0.0029439207632094623, |
| "step": 1520, |
| "step_time": 7.905811912985518 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 566.5, |
| "completions/max_terminated_length": 566.5, |
| "completions/mean_length": 317.6875, |
| "completions/mean_terminated_length": 317.6875, |
| "completions/min_length": 197.4, |
| "completions/min_terminated_length": 197.4, |
| "entropy": 0.06675491034984589, |
| "epoch": 1.6381156316916488, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.847100000000001e-06, |
| "loss": 0.0013, |
| "num_tokens": 93168373.0, |
| "reward": 1.056250023841858, |
| "reward_std": 0.136273393034935, |
| "rewards/reward_accuracy/mean": 0.95625, |
| "rewards/reward_accuracy/std": 0.1362733945250511, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8859559655189515, |
| "sampling/importance_sampling_ratio/mean": 1.0007485449314117, |
| "sampling/importance_sampling_ratio/min": 0.507869890332222, |
| "sampling/sampling_logp_difference/max": 0.3700533539056778, |
| "sampling/sampling_logp_difference/mean": 0.0027073199627920983, |
| "step": 1530, |
| "step_time": 7.504971977323294 |
| }, |
| { |
| "clip_ratio/high_max": 2.212316685472615e-05, |
| "clip_ratio/high_mean": 5.5307917136815375e-06, |
| "clip_ratio/low_mean": 2.8722426577587614e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 8.403034371440299e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 546.9, |
| "completions/max_terminated_length": 546.9, |
| "completions/mean_length": 312.2203125, |
| "completions/mean_terminated_length": 312.2203125, |
| "completions/min_length": 188.9, |
| "completions/min_terminated_length": 188.9, |
| "entropy": 0.07487013121135533, |
| "epoch": 1.6488222698072805, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.349609375, |
| "learning_rate": 9.8461e-06, |
| "loss": 0.0033, |
| "num_tokens": 93499218.0, |
| "reward": 1.0656250238418579, |
| "reward_std": 0.16519061625003814, |
| "rewards/reward_accuracy/mean": 0.965625, |
| "rewards/reward_accuracy/std": 0.1651906192302704, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9226843595504761, |
| "sampling/importance_sampling_ratio/mean": 1.0024244129657744, |
| "sampling/importance_sampling_ratio/min": 0.48585228924639523, |
| "sampling/sampling_logp_difference/max": 0.861660385131836, |
| "sampling/sampling_logp_difference/mean": 0.002868760307319462, |
| "step": 1540, |
| "step_time": 7.075284147867933 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 544.0, |
| "completions/max_terminated_length": 544.0, |
| "completions/mean_length": 304.8078125, |
| "completions/mean_terminated_length": 304.8078125, |
| "completions/min_length": 206.1, |
| "completions/min_terminated_length": 206.1, |
| "entropy": 0.07690851246006787, |
| "epoch": 1.6595289079229123, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.48046875, |
| "learning_rate": 9.8451e-06, |
| "loss": 0.007, |
| "num_tokens": 93822647.0, |
| "reward": 1.0578125238418579, |
| "reward_std": 0.17215676605701447, |
| "rewards/reward_accuracy/mean": 0.9578125, |
| "rewards/reward_accuracy/std": 0.17215677350759506, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9077085256576538, |
| "sampling/importance_sampling_ratio/mean": 1.003688645362854, |
| "sampling/importance_sampling_ratio/min": 0.4427179664373398, |
| "sampling/sampling_logp_difference/max": 0.7035780668258667, |
| "sampling/sampling_logp_difference/mean": 0.003030651854351163, |
| "step": 1550, |
| "step_time": 7.18392994645983 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 508.5, |
| "completions/max_terminated_length": 508.5, |
| "completions/mean_length": 316.20625, |
| "completions/mean_terminated_length": 316.20625, |
| "completions/min_length": 198.5, |
| "completions/min_terminated_length": 198.5, |
| "entropy": 0.06954822610132397, |
| "epoch": 1.6702355460385439, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.25390625, |
| "learning_rate": 9.844100000000001e-06, |
| "loss": 0.0042, |
| "num_tokens": 94152811.0, |
| "reward": 1.092187523841858, |
| "reward_std": 0.055036810040473935, |
| "rewards/reward_accuracy/mean": 0.9921875, |
| "rewards/reward_accuracy/std": 0.055036810040473935, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7852643847465515, |
| "sampling/importance_sampling_ratio/mean": 1.0035234212875366, |
| "sampling/importance_sampling_ratio/min": 0.45582168996334077, |
| "sampling/sampling_logp_difference/max": 0.6839189767837525, |
| "sampling/sampling_logp_difference/mean": 0.0026874721050262453, |
| "step": 1560, |
| "step_time": 6.759840969089419 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 659.8, |
| "completions/max_terminated_length": 659.8, |
| "completions/mean_length": 322.934375, |
| "completions/mean_terminated_length": 322.934375, |
| "completions/min_length": 203.8, |
| "completions/min_terminated_length": 203.8, |
| "entropy": 0.06754827189724892, |
| "epoch": 1.6809421841541756, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.333984375, |
| "learning_rate": 9.8431e-06, |
| "loss": 0.0043, |
| "num_tokens": 94487457.0, |
| "reward": 1.064062523841858, |
| "reward_std": 0.1390695095062256, |
| "rewards/reward_accuracy/mean": 0.9640625, |
| "rewards/reward_accuracy/std": 0.1390695095062256, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0658219337463377, |
| "sampling/importance_sampling_ratio/mean": 1.003229957818985, |
| "sampling/importance_sampling_ratio/min": 0.4485538862645626, |
| "sampling/sampling_logp_difference/max": 0.7008833765983582, |
| "sampling/sampling_logp_difference/mean": 0.002741540456190705, |
| "step": 1570, |
| "step_time": 8.128997852792963 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 6.387980192812392e-07, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.387980192812392e-07, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1072.1, |
| "completions/max_terminated_length": 1072.1, |
| "completions/mean_length": 354.86875, |
| "completions/mean_terminated_length": 354.86875, |
| "completions/min_length": 203.6, |
| "completions/min_terminated_length": 203.6, |
| "entropy": 0.08077807566151023, |
| "epoch": 1.6916488222698072, |
| "frac_reward_zero_std": 0.75, |
| "grad_norm": 0.494140625, |
| "learning_rate": 9.842100000000002e-06, |
| "loss": -0.0031, |
| "num_tokens": 94844061.0, |
| "reward": 0.9982031464576722, |
| "reward_std": 0.26102162301540377, |
| "rewards/reward_accuracy/mean": 0.8984375, |
| "rewards/reward_accuracy/std": 0.26040040850639345, |
| "rewards/reward_format/mean": 0.0997656263411045, |
| "rewards/reward_format/std": 0.0013886408880352974, |
| "sampling/importance_sampling_ratio/max": 2.053833174705505, |
| "sampling/importance_sampling_ratio/mean": 1.0073238134384155, |
| "sampling/importance_sampling_ratio/min": 0.38636885648593305, |
| "sampling/sampling_logp_difference/max": 0.860472047328949, |
| "sampling/sampling_logp_difference/mean": 0.0032895981101319196, |
| "step": 1580, |
| "step_time": 12.440732840960845 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 820.2, |
| "completions/max_terminated_length": 592.0, |
| "completions/mean_length": 358.190625, |
| "completions/mean_terminated_length": 354.03966064453124, |
| "completions/min_length": 222.8, |
| "completions/min_terminated_length": 222.8, |
| "entropy": 0.07479010678362101, |
| "epoch": 1.702355460385439, |
| "frac_reward_zero_std": 0.7625, |
| "grad_norm": 0.46875, |
| "learning_rate": 9.841100000000001e-06, |
| "loss": 0.0074, |
| "num_tokens": 95202103.0, |
| "reward": 1.0123437702655793, |
| "reward_std": 0.22732870578765868, |
| "rewards/reward_accuracy/mean": 0.9125, |
| "rewards/reward_accuracy/std": 0.22694342732429504, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 1.8977302074432374, |
| "sampling/importance_sampling_ratio/mean": 1.0068412601947785, |
| "sampling/importance_sampling_ratio/min": 0.4609717845916748, |
| "sampling/sampling_logp_difference/max": 0.4274681627750397, |
| "sampling/sampling_logp_difference/mean": 0.0030843788292258976, |
| "step": 1590, |
| "step_time": 9.698968959227205 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 705.3, |
| "completions/max_terminated_length": 705.3, |
| "completions/mean_length": 383.69375, |
| "completions/mean_terminated_length": 383.69375, |
| "completions/min_length": 224.4, |
| "completions/min_terminated_length": 224.4, |
| "entropy": 0.0746586648048833, |
| "epoch": 1.7130620985010707, |
| "frac_reward_zero_std": 0.9375, |
| "grad_norm": 0.0, |
| "learning_rate": 9.840100000000001e-06, |
| "loss": -0.0011, |
| "num_tokens": 95580227.0, |
| "reward": 1.059375023841858, |
| "reward_std": 0.12984935343265533, |
| "rewards/reward_accuracy/mean": 0.959375, |
| "rewards/reward_accuracy/std": 0.12984935492277144, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.1005443334579468, |
| "sampling/importance_sampling_ratio/mean": 1.0024439632892608, |
| "sampling/importance_sampling_ratio/min": 0.38014114946126937, |
| "sampling/sampling_logp_difference/max": 0.46549739837646487, |
| "sampling/sampling_logp_difference/mean": 0.003083315584808588, |
| "step": 1600, |
| "step_time": 8.588696904014796 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 725.2, |
| "completions/max_terminated_length": 725.2, |
| "completions/mean_length": 361.0765625, |
| "completions/mean_terminated_length": 361.0765625, |
| "completions/min_length": 230.4, |
| "completions/min_terminated_length": 230.4, |
| "entropy": 0.08019938471261412, |
| "epoch": 1.7237687366167025, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8391e-06, |
| "loss": -0.0055, |
| "num_tokens": 95939188.0, |
| "reward": 1.079687523841858, |
| "reward_std": 0.08525067269802093, |
| "rewards/reward_accuracy/mean": 0.9796875, |
| "rewards/reward_accuracy/std": 0.08525067865848542, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.058314311504364, |
| "sampling/importance_sampling_ratio/mean": 0.998687607049942, |
| "sampling/importance_sampling_ratio/min": 0.452979052066803, |
| "sampling/sampling_logp_difference/max": 0.4121090054512024, |
| "sampling/sampling_logp_difference/mean": 0.003351893392391503, |
| "step": 1610, |
| "step_time": 8.774539006268606 |
| }, |
| { |
| "clip_ratio/high_max": 3.0206462542992086e-05, |
| "clip_ratio/high_mean": 7.5516156357480215e-06, |
| "clip_ratio/low_mean": 2.4841017875587566e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.0035717423306778e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 812.7, |
| "completions/max_terminated_length": 812.7, |
| "completions/mean_length": 376.8875, |
| "completions/mean_terminated_length": 376.8875, |
| "completions/min_length": 209.8, |
| "completions/min_terminated_length": 209.8, |
| "entropy": 0.09234049175865948, |
| "epoch": 1.734475374732334, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.66015625, |
| "learning_rate": 9.8381e-06, |
| "loss": 0.0024, |
| "num_tokens": 96310252.0, |
| "reward": 0.985937523841858, |
| "reward_std": 0.30762080252170565, |
| "rewards/reward_accuracy/mean": 0.8859375, |
| "rewards/reward_accuracy/std": 0.30762080997228625, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0571969151496887, |
| "sampling/importance_sampling_ratio/mean": 0.9925124883651734, |
| "sampling/importance_sampling_ratio/min": 0.3672985717654228, |
| "sampling/sampling_logp_difference/max": 0.4966865718364716, |
| "sampling/sampling_logp_difference/mean": 0.0038217913126572965, |
| "step": 1620, |
| "step_time": 9.613462931942195 |
| }, |
| { |
| "clip_ratio/high_max": 7.87153621786274e-06, |
| "clip_ratio/high_mean": 1.967884054465685e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.967884054465685e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 837.1, |
| "completions/max_terminated_length": 837.1, |
| "completions/mean_length": 401.8515625, |
| "completions/mean_terminated_length": 401.8515625, |
| "completions/min_length": 249.1, |
| "completions/min_terminated_length": 249.1, |
| "entropy": 0.09493691669777035, |
| "epoch": 1.7451820128479656, |
| "frac_reward_zero_std": 0.7125, |
| "grad_norm": 0.2890625, |
| "learning_rate": 9.837100000000001e-06, |
| "loss": -0.0014, |
| "num_tokens": 96696253.0, |
| "reward": 0.9609375238418579, |
| "reward_std": 0.32564790844917296, |
| "rewards/reward_accuracy/mean": 0.8609375, |
| "rewards/reward_accuracy/std": 0.3256479248404503, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0345356106758117, |
| "sampling/importance_sampling_ratio/mean": 1.0005013525485993, |
| "sampling/importance_sampling_ratio/min": 0.3664106011390686, |
| "sampling/sampling_logp_difference/max": 0.4841954469680786, |
| "sampling/sampling_logp_difference/mean": 0.0037874059984460474, |
| "step": 1630, |
| "step_time": 9.691187618020923 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.8189755792263896e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.8189755792263896e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 823.7, |
| "completions/max_terminated_length": 823.7, |
| "completions/mean_length": 386.8515625, |
| "completions/mean_terminated_length": 386.8515625, |
| "completions/min_length": 241.9, |
| "completions/min_terminated_length": 241.9, |
| "entropy": 0.10402072858996689, |
| "epoch": 1.7558886509635974, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.357421875, |
| "learning_rate": 9.8361e-06, |
| "loss": 0.0081, |
| "num_tokens": 97070638.0, |
| "reward": 1.025000023841858, |
| "reward_std": 0.228138467669487, |
| "rewards/reward_accuracy/mean": 0.925, |
| "rewards/reward_accuracy/std": 0.22813847064971923, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0830006122589113, |
| "sampling/importance_sampling_ratio/mean": 0.9828476965427398, |
| "sampling/importance_sampling_ratio/min": 0.3422037959098816, |
| "sampling/sampling_logp_difference/max": 0.4761467456817627, |
| "sampling/sampling_logp_difference/mean": 0.003937892033718526, |
| "step": 1640, |
| "step_time": 9.496324560185894 |
| }, |
| { |
| "clip_ratio/high_max": 9.245562250725924e-06, |
| "clip_ratio/high_mean": 2.311390562681481e-06, |
| "clip_ratio/low_mean": 8.709587746125181e-07, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.1823493372939994e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 757.6, |
| "completions/max_terminated_length": 757.6, |
| "completions/mean_length": 392.340625, |
| "completions/mean_terminated_length": 392.340625, |
| "completions/min_length": 254.0, |
| "completions/min_terminated_length": 254.0, |
| "entropy": 0.10269364658743144, |
| "epoch": 1.7665952890792291, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.146484375, |
| "learning_rate": 9.8351e-06, |
| "loss": -0.0038, |
| "num_tokens": 97452168.0, |
| "reward": 1.0421875238418579, |
| "reward_std": 0.18650540411472322, |
| "rewards/reward_accuracy/mean": 0.9421875, |
| "rewards/reward_accuracy/std": 0.18650540709495544, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0714402556419373, |
| "sampling/importance_sampling_ratio/mean": 0.9962391078472137, |
| "sampling/importance_sampling_ratio/min": 0.4453423172235489, |
| "sampling/sampling_logp_difference/max": 0.3763102412223816, |
| "sampling/sampling_logp_difference/mean": 0.0037064227042719724, |
| "step": 1650, |
| "step_time": 8.989678170578554 |
| }, |
| { |
| "clip_ratio/high_max": 1.4350117635331116e-05, |
| "clip_ratio/high_mean": 3.587529408832779e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.587529408832779e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 723.7, |
| "completions/max_terminated_length": 723.7, |
| "completions/mean_length": 404.5859375, |
| "completions/mean_terminated_length": 404.5859375, |
| "completions/min_length": 264.0, |
| "completions/min_terminated_length": 264.0, |
| "entropy": 0.12140203267335892, |
| "epoch": 1.777301927194861, |
| "frac_reward_zero_std": 0.9, |
| "grad_norm": 0.0, |
| "learning_rate": 9.834100000000001e-06, |
| "loss": 0.0075, |
| "num_tokens": 97842655.0, |
| "reward": 1.064062523841858, |
| "reward_std": 0.14970277547836303, |
| "rewards/reward_accuracy/mean": 0.9640625, |
| "rewards/reward_accuracy/std": 0.1497027814388275, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.172510254383087, |
| "sampling/importance_sampling_ratio/mean": 0.9883950352668762, |
| "sampling/importance_sampling_ratio/min": 0.3636951893568039, |
| "sampling/sampling_logp_difference/max": 0.37829705476760866, |
| "sampling/sampling_logp_difference/mean": 0.0042060357984155415, |
| "step": 1660, |
| "step_time": 8.666362385451794 |
| }, |
| { |
| "clip_ratio/high_max": 4.678143886849284e-06, |
| "clip_ratio/high_mean": 1.169535971712321e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.169535971712321e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1010.6, |
| "completions/max_terminated_length": 1010.6, |
| "completions/mean_length": 447.575, |
| "completions/mean_terminated_length": 447.575, |
| "completions/min_length": 275.5, |
| "completions/min_terminated_length": 275.5, |
| "entropy": 0.12740630861371754, |
| "epoch": 1.7880085653104925, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.376953125, |
| "learning_rate": 9.8331e-06, |
| "loss": -0.0023, |
| "num_tokens": 98257503.0, |
| "reward": 1.0500000238418579, |
| "reward_std": 0.1851152718067169, |
| "rewards/reward_accuracy/mean": 0.95, |
| "rewards/reward_accuracy/std": 0.1851152777671814, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.203228497505188, |
| "sampling/importance_sampling_ratio/mean": 1.0162541031837464, |
| "sampling/importance_sampling_ratio/min": 0.3328179895877838, |
| "sampling/sampling_logp_difference/max": 0.4627582848072052, |
| "sampling/sampling_logp_difference/mean": 0.004239224758930504, |
| "step": 1670, |
| "step_time": 11.33724239524454 |
| }, |
| { |
| "clip_ratio/high_max": 6.038647188688628e-06, |
| "clip_ratio/high_mean": 1.509661797172157e-06, |
| "clip_ratio/low_mean": 1.553180845803581e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.0628426429757384e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 812.5, |
| "completions/max_terminated_length": 812.5, |
| "completions/mean_length": 491.4453125, |
| "completions/mean_terminated_length": 491.4453125, |
| "completions/min_length": 268.9, |
| "completions/min_terminated_length": 268.9, |
| "entropy": 0.14627750939689577, |
| "epoch": 1.7987152034261242, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.423828125, |
| "learning_rate": 9.8321e-06, |
| "loss": 0.0026, |
| "num_tokens": 98703388.0, |
| "reward": 1.019921898841858, |
| "reward_std": 0.22838517278432846, |
| "rewards/reward_accuracy/mean": 0.9203125, |
| "rewards/reward_accuracy/std": 0.22626071125268937, |
| "rewards/reward_format/mean": 0.09960937723517418, |
| "rewards/reward_format/std": 0.0031250000931322573, |
| "sampling/importance_sampling_ratio/max": 2.342504560947418, |
| "sampling/importance_sampling_ratio/mean": 1.00094313621521, |
| "sampling/importance_sampling_ratio/min": 0.2615745931863785, |
| "sampling/sampling_logp_difference/max": 0.38652130365371706, |
| "sampling/sampling_logp_difference/mean": 0.004804426850751042, |
| "step": 1680, |
| "step_time": 9.53578919605352 |
| }, |
| { |
| "clip_ratio/high_max": 2.575177641119808e-05, |
| "clip_ratio/high_mean": 6.43794410279952e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.43794410279952e-06, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 1198.3, |
| "completions/max_terminated_length": 978.2, |
| "completions/mean_length": 467.2453125, |
| "completions/mean_terminated_length": 459.3212860107422, |
| "completions/min_length": 269.0, |
| "completions/min_terminated_length": 269.0, |
| "entropy": 0.11198789249174297, |
| "epoch": 1.8094218415417558, |
| "frac_reward_zero_std": 0.7625, |
| "grad_norm": 0.49609375, |
| "learning_rate": 9.831100000000001e-06, |
| "loss": 0.0108, |
| "num_tokens": 99133801.0, |
| "reward": 1.016875022649765, |
| "reward_std": 0.2391919732093811, |
| "rewards/reward_accuracy/mean": 0.9171875, |
| "rewards/reward_accuracy/std": 0.2384605273604393, |
| "rewards/reward_format/mean": 0.09968750178813934, |
| "rewards/reward_format/std": 0.002500000037252903, |
| "sampling/importance_sampling_ratio/max": 2.347457456588745, |
| "sampling/importance_sampling_ratio/mean": 1.0278074145317078, |
| "sampling/importance_sampling_ratio/min": 0.34849384129047395, |
| "sampling/sampling_logp_difference/max": 0.5231088638305664, |
| "sampling/sampling_logp_difference/mean": 0.004168810788542032, |
| "step": 1690, |
| "step_time": 13.347568203089759 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 1188.3, |
| "completions/max_terminated_length": 940.6, |
| "completions/mean_length": 401.5078125, |
| "completions/mean_terminated_length": 397.299853515625, |
| "completions/min_length": 237.8, |
| "completions/min_terminated_length": 237.8, |
| "entropy": 0.1036820865701884, |
| "epoch": 1.8201284796573876, |
| "frac_reward_zero_std": 0.925, |
| "grad_norm": 0.302734375, |
| "learning_rate": 9.830100000000001e-06, |
| "loss": 0.0017, |
| "num_tokens": 99519326.0, |
| "reward": 1.0092187762260436, |
| "reward_std": 0.2331274539232254, |
| "rewards/reward_accuracy/mean": 0.909375, |
| "rewards/reward_accuracy/std": 0.2323932930827141, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 2.117543613910675, |
| "sampling/importance_sampling_ratio/mean": 0.9944232642650604, |
| "sampling/importance_sampling_ratio/min": 0.22871265858411788, |
| "sampling/sampling_logp_difference/max": 0.4097829580307007, |
| "sampling/sampling_logp_difference/mean": 0.004051772458478808, |
| "step": 1700, |
| "step_time": 12.778053979808465 |
| }, |
| { |
| "clip_ratio/high_max": 9.441087604500353e-06, |
| "clip_ratio/high_mean": 2.3602719011250882e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.3602719011250882e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 986.1, |
| "completions/max_terminated_length": 986.1, |
| "completions/mean_length": 442.5328125, |
| "completions/mean_terminated_length": 442.5328125, |
| "completions/min_length": 260.7, |
| "completions/min_terminated_length": 260.7, |
| "entropy": 0.09147540256381034, |
| "epoch": 1.8308351177730193, |
| "frac_reward_zero_std": 0.85, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8291e-06, |
| "loss": -0.0083, |
| "num_tokens": 99933347.0, |
| "reward": 1.021875023841858, |
| "reward_std": 0.21622758209705353, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.21622758209705353, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.2068938493728636, |
| "sampling/importance_sampling_ratio/mean": 0.9750546932220459, |
| "sampling/importance_sampling_ratio/min": 0.29905890859663486, |
| "sampling/sampling_logp_difference/max": 0.7923729062080384, |
| "sampling/sampling_logp_difference/mean": 0.003699657041579485, |
| "step": 1710, |
| "step_time": 11.114837915683164 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 715.0, |
| "completions/max_terminated_length": 715.0, |
| "completions/mean_length": 416.775, |
| "completions/mean_terminated_length": 416.775, |
| "completions/min_length": 260.1, |
| "completions/min_terminated_length": 260.1, |
| "entropy": 0.09530194201506674, |
| "epoch": 1.841541755888651, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.4140625, |
| "learning_rate": 9.8281e-06, |
| "loss": -0.0044, |
| "num_tokens": 100329683.0, |
| "reward": 1.0592187762260437, |
| "reward_std": 0.17502699196338653, |
| "rewards/reward_accuracy/mean": 0.959375, |
| "rewards/reward_accuracy/std": 0.17461777478456497, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 1.917730438709259, |
| "sampling/importance_sampling_ratio/mean": 0.9825216412544251, |
| "sampling/importance_sampling_ratio/min": 0.3752861022949219, |
| "sampling/sampling_logp_difference/max": 0.47720618844032286, |
| "sampling/sampling_logp_difference/mean": 0.0037107625510543587, |
| "step": 1720, |
| "step_time": 8.69084729538299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 657.8, |
| "completions/max_terminated_length": 657.8, |
| "completions/mean_length": 422.0, |
| "completions/mean_terminated_length": 422.0, |
| "completions/min_length": 265.8, |
| "completions/min_terminated_length": 265.8, |
| "entropy": 0.09505298319272697, |
| "epoch": 1.8522483940042827, |
| "frac_reward_zero_std": 0.95, |
| "grad_norm": 0.18359375, |
| "learning_rate": 9.827100000000001e-06, |
| "loss": -0.0034, |
| "num_tokens": 100730467.0, |
| "reward": 1.068750023841858, |
| "reward_std": 0.11236459910869598, |
| "rewards/reward_accuracy/mean": 0.96875, |
| "rewards/reward_accuracy/std": 0.11236459910869598, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0214869141578675, |
| "sampling/importance_sampling_ratio/mean": 1.0046520054340362, |
| "sampling/importance_sampling_ratio/min": 0.41623824238777163, |
| "sampling/sampling_logp_difference/max": 0.3673498034477234, |
| "sampling/sampling_logp_difference/mean": 0.0036344136809930206, |
| "step": 1730, |
| "step_time": 8.152937957458198 |
| }, |
| { |
| "clip_ratio/high_max": 1.059322021319531e-05, |
| "clip_ratio/high_mean": 2.6483050532988274e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.6483050532988274e-06, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 940.0, |
| "completions/max_terminated_length": 695.4, |
| "completions/mean_length": 404.5296875, |
| "completions/mean_terminated_length": 400.36083984375, |
| "completions/min_length": 257.5, |
| "completions/min_terminated_length": 257.5, |
| "entropy": 0.10102281142026186, |
| "epoch": 1.8629550321199142, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.36328125, |
| "learning_rate": 9.8261e-06, |
| "loss": -0.0069, |
| "num_tokens": 101117030.0, |
| "reward": 1.0514062762260437, |
| "reward_std": 0.17088834047317505, |
| "rewards/reward_accuracy/mean": 0.9515625, |
| "rewards/reward_accuracy/std": 0.17045110911130906, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 1.8757730841636657, |
| "sampling/importance_sampling_ratio/mean": 0.9968436300754547, |
| "sampling/importance_sampling_ratio/min": 0.37131072729825976, |
| "sampling/sampling_logp_difference/max": 0.45754934549331666, |
| "sampling/sampling_logp_difference/mean": 0.003871224564500153, |
| "step": 1740, |
| "step_time": 10.6782959821634 |
| }, |
| { |
| "clip_ratio/high_max": 2.791573278955184e-05, |
| "clip_ratio/high_mean": 6.97893319738796e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.97893319738796e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 660.0, |
| "completions/max_terminated_length": 660.0, |
| "completions/mean_length": 428.7125, |
| "completions/mean_terminated_length": 428.7125, |
| "completions/min_length": 291.4, |
| "completions/min_terminated_length": 291.4, |
| "entropy": 0.09220113810151816, |
| "epoch": 1.873661670235546, |
| "frac_reward_zero_std": 0.9, |
| "grad_norm": 0.373046875, |
| "learning_rate": 9.8251e-06, |
| "loss": -0.0013, |
| "num_tokens": 101520318.0, |
| "reward": 1.0546875238418578, |
| "reward_std": 0.1736715942621231, |
| "rewards/reward_accuracy/mean": 0.9546875, |
| "rewards/reward_accuracy/std": 0.17367159575223923, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.1409215331077576, |
| "sampling/importance_sampling_ratio/mean": 0.9904058933258056, |
| "sampling/importance_sampling_ratio/min": 0.37598259150981905, |
| "sampling/sampling_logp_difference/max": 0.3789864182472229, |
| "sampling/sampling_logp_difference/mean": 0.003363468893803656, |
| "step": 1750, |
| "step_time": 8.196429524570704 |
| }, |
| { |
| "clip_ratio/high_max": 9.286775457439944e-06, |
| "clip_ratio/high_mean": 2.321693864359986e-06, |
| "clip_ratio/low_mean": 2.7412281269789675e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 5.062921991338953e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 822.4, |
| "completions/max_terminated_length": 822.4, |
| "completions/mean_length": 443.2828125, |
| "completions/mean_terminated_length": 443.2828125, |
| "completions/min_length": 260.4, |
| "completions/min_terminated_length": 260.4, |
| "entropy": 0.09848947180435061, |
| "epoch": 1.8843683083511777, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.25390625, |
| "learning_rate": 9.824100000000001e-06, |
| "loss": 0.0096, |
| "num_tokens": 101933379.0, |
| "reward": 1.035937523841858, |
| "reward_std": 0.19951338469982147, |
| "rewards/reward_accuracy/mean": 0.9359375, |
| "rewards/reward_accuracy/std": 0.19951339215040206, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.146065080165863, |
| "sampling/importance_sampling_ratio/mean": 0.9994051575660705, |
| "sampling/importance_sampling_ratio/min": 0.39188658595085146, |
| "sampling/sampling_logp_difference/max": 0.4163475394248962, |
| "sampling/sampling_logp_difference/mean": 0.0035301547264680266, |
| "step": 1760, |
| "step_time": 9.737725848052651 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 781.2, |
| "completions/max_terminated_length": 781.2, |
| "completions/mean_length": 431.46875, |
| "completions/mean_terminated_length": 431.46875, |
| "completions/min_length": 254.8, |
| "completions/min_terminated_length": 254.8, |
| "entropy": 0.10197207778692245, |
| "epoch": 1.8950749464668095, |
| "frac_reward_zero_std": 0.9125, |
| "grad_norm": 0.26953125, |
| "learning_rate": 9.823100000000001e-06, |
| "loss": -0.0001, |
| "num_tokens": 102341935.0, |
| "reward": 1.021875023841858, |
| "reward_std": 0.2387033075094223, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.2387033149600029, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8940268158912659, |
| "sampling/importance_sampling_ratio/mean": 0.9708118796348572, |
| "sampling/importance_sampling_ratio/min": 0.4274204343557358, |
| "sampling/sampling_logp_difference/max": 0.3806558132171631, |
| "sampling/sampling_logp_difference/mean": 0.0035763060208410026, |
| "step": 1770, |
| "step_time": 9.16180561692454 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 966.7, |
| "completions/max_terminated_length": 966.7, |
| "completions/mean_length": 394.55625, |
| "completions/mean_terminated_length": 394.55625, |
| "completions/min_length": 247.8, |
| "completions/min_terminated_length": 247.8, |
| "entropy": 0.09909435922745616, |
| "epoch": 1.905781584582441, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8221e-06, |
| "loss": -0.0024, |
| "num_tokens": 102723811.0, |
| "reward": 1.0421875238418579, |
| "reward_std": 0.16203955709934234, |
| "rewards/reward_accuracy/mean": 0.9421875, |
| "rewards/reward_accuracy/std": 0.16203955858945845, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9555251836776733, |
| "sampling/importance_sampling_ratio/mean": 0.9755014598369598, |
| "sampling/importance_sampling_ratio/min": 0.4305851817131042, |
| "sampling/sampling_logp_difference/max": 0.3780761957168579, |
| "sampling/sampling_logp_difference/mean": 0.0035153797827661036, |
| "step": 1780, |
| "step_time": 10.97240506974049 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 671.0, |
| "completions/max_terminated_length": 671.0, |
| "completions/mean_length": 407.590625, |
| "completions/mean_terminated_length": 407.590625, |
| "completions/min_length": 243.9, |
| "completions/min_terminated_length": 243.9, |
| "entropy": 0.10056707290932536, |
| "epoch": 1.9164882226980728, |
| "frac_reward_zero_std": 0.7625, |
| "grad_norm": 0.37890625, |
| "learning_rate": 9.821100000000002e-06, |
| "loss": -0.0026, |
| "num_tokens": 103115933.0, |
| "reward": 1.037500023841858, |
| "reward_std": 0.20822631120681762, |
| "rewards/reward_accuracy/mean": 0.9375, |
| "rewards/reward_accuracy/std": 0.20822631567716599, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9552023649215697, |
| "sampling/importance_sampling_ratio/mean": 1.0122333884239196, |
| "sampling/importance_sampling_ratio/min": 0.44874298572540283, |
| "sampling/sampling_logp_difference/max": 0.34589052200317383, |
| "sampling/sampling_logp_difference/mean": 0.003413649625144899, |
| "step": 1790, |
| "step_time": 8.214828142989427 |
| }, |
| { |
| "clip_ratio/high_max": 7.521059160353616e-06, |
| "clip_ratio/high_mean": 1.880264790088404e-06, |
| "clip_ratio/low_mean": 2.499999936844688e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.380264726933092e-06, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 1183.1, |
| "completions/max_terminated_length": 952.5, |
| "completions/mean_length": 412.8875, |
| "completions/mean_terminated_length": 396.4402282714844, |
| "completions/min_length": 231.7, |
| "completions/min_terminated_length": 231.7, |
| "entropy": 0.11350611159577965, |
| "epoch": 1.9271948608137044, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.820100000000001e-06, |
| "loss": 0.0006, |
| "num_tokens": 103504645.0, |
| "reward": 1.0337500274181366, |
| "reward_std": 0.22465506196022034, |
| "rewards/reward_accuracy/mean": 0.934375, |
| "rewards/reward_accuracy/std": 0.22287926226854324, |
| "rewards/reward_format/mean": 0.09937500208616257, |
| "rewards/reward_format/std": 0.0033804203383624555, |
| "sampling/importance_sampling_ratio/max": 1.8912512898445129, |
| "sampling/importance_sampling_ratio/mean": 0.9944418668746948, |
| "sampling/importance_sampling_ratio/min": 0.3780899077653885, |
| "sampling/sampling_logp_difference/max": 0.3172037720680237, |
| "sampling/sampling_logp_difference/mean": 0.0035695731872692703, |
| "step": 1800, |
| "step_time": 13.236138972593471 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 1210.2, |
| "completions/max_terminated_length": 780.6, |
| "completions/mean_length": 432.7921875, |
| "completions/mean_terminated_length": 424.56878051757815, |
| "completions/min_length": 254.0, |
| "completions/min_terminated_length": 254.0, |
| "entropy": 0.11868407726287841, |
| "epoch": 1.9379014989293362, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.50390625, |
| "learning_rate": 9.8191e-06, |
| "loss": -0.0101, |
| "num_tokens": 103914160.0, |
| "reward": 1.0371875286102294, |
| "reward_std": 0.1841567039489746, |
| "rewards/reward_accuracy/mean": 0.9375, |
| "rewards/reward_accuracy/std": 0.183076374232769, |
| "rewards/reward_format/mean": 0.09968750178813934, |
| "rewards/reward_format/std": 0.002500000037252903, |
| "sampling/importance_sampling_ratio/max": 1.9401855230331422, |
| "sampling/importance_sampling_ratio/mean": 0.9919736981391907, |
| "sampling/importance_sampling_ratio/min": 0.4217506214976311, |
| "sampling/sampling_logp_difference/max": 0.33649542927742004, |
| "sampling/sampling_logp_difference/mean": 0.0037801675265654922, |
| "step": 1810, |
| "step_time": 13.282955834129826 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 827.6, |
| "completions/max_terminated_length": 827.6, |
| "completions/mean_length": 420.10625, |
| "completions/mean_terminated_length": 420.10625, |
| "completions/min_length": 257.1, |
| "completions/min_terminated_length": 257.1, |
| "entropy": 0.11690366845577956, |
| "epoch": 1.948608137044968, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8181e-06, |
| "loss": -0.0007, |
| "num_tokens": 104313124.0, |
| "reward": 1.053125023841858, |
| "reward_std": 0.1910089522600174, |
| "rewards/reward_accuracy/mean": 0.953125, |
| "rewards/reward_accuracy/std": 0.19100895673036575, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.012774312496185, |
| "sampling/importance_sampling_ratio/mean": 1.0011790215969085, |
| "sampling/importance_sampling_ratio/min": 0.34246390759944917, |
| "sampling/sampling_logp_difference/max": 0.31694276332855226, |
| "sampling/sampling_logp_difference/mean": 0.003591471561230719, |
| "step": 1820, |
| "step_time": 9.540361557714641 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 5.5364751460729165e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 5.5364751460729165e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 754.0, |
| "completions/max_terminated_length": 754.0, |
| "completions/mean_length": 389.16875, |
| "completions/mean_terminated_length": 389.16875, |
| "completions/min_length": 248.3, |
| "completions/min_terminated_length": 248.3, |
| "entropy": 0.13287138412706553, |
| "epoch": 1.9593147751605997, |
| "frac_reward_zero_std": 0.7625, |
| "grad_norm": 0.0, |
| "learning_rate": 9.817100000000001e-06, |
| "loss": -0.0063, |
| "num_tokens": 104693136.0, |
| "reward": 1.051562523841858, |
| "reward_std": 0.18550011813640593, |
| "rewards/reward_accuracy/mean": 0.9515625, |
| "rewards/reward_accuracy/std": 0.18550012707710267, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9615930438041687, |
| "sampling/importance_sampling_ratio/mean": 0.9950068831443787, |
| "sampling/importance_sampling_ratio/min": 0.390790793299675, |
| "sampling/sampling_logp_difference/max": 0.3105659455060959, |
| "sampling/sampling_logp_difference/mean": 0.003926422935910523, |
| "step": 1830, |
| "step_time": 8.939409441500903 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 877.9, |
| "completions/max_terminated_length": 877.9, |
| "completions/mean_length": 403.653125, |
| "completions/mean_terminated_length": 403.653125, |
| "completions/min_length": 240.3, |
| "completions/min_terminated_length": 240.3, |
| "entropy": 0.10936231981031597, |
| "epoch": 1.9700214132762313, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.48828125, |
| "learning_rate": 9.8161e-06, |
| "loss": 0.0017, |
| "num_tokens": 105076802.0, |
| "reward": 1.012500023841858, |
| "reward_std": 0.25887428820133207, |
| "rewards/reward_accuracy/mean": 0.9125, |
| "rewards/reward_accuracy/std": 0.25887429118156435, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.105149734020233, |
| "sampling/importance_sampling_ratio/mean": 0.9928551197052002, |
| "sampling/importance_sampling_ratio/min": 0.4438862532377243, |
| "sampling/sampling_logp_difference/max": 0.36968438923358915, |
| "sampling/sampling_logp_difference/mean": 0.0034039631020277737, |
| "step": 1840, |
| "step_time": 10.190516875032335 |
| }, |
| { |
| "clip_ratio/high_max": 7.512019510613754e-06, |
| "clip_ratio/high_mean": 1.8780048776534386e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.8780048776534386e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 790.5, |
| "completions/max_terminated_length": 790.5, |
| "completions/mean_length": 414.2171875, |
| "completions/mean_terminated_length": 414.2171875, |
| "completions/min_length": 272.0, |
| "completions/min_terminated_length": 272.0, |
| "entropy": 0.11457934873178602, |
| "epoch": 1.9807280513918628, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8151e-06, |
| "loss": 0.0109, |
| "num_tokens": 105473117.0, |
| "reward": 1.0529687881469727, |
| "reward_std": 0.17824327945709229, |
| "rewards/reward_accuracy/mean": 0.953125, |
| "rewards/reward_accuracy/std": 0.17825458496809005, |
| "rewards/reward_format/mean": 0.09984375238418579, |
| "rewards/reward_format/std": 0.0012500001117587089, |
| "sampling/importance_sampling_ratio/max": 1.871982204914093, |
| "sampling/importance_sampling_ratio/mean": 0.9994886100292206, |
| "sampling/importance_sampling_ratio/min": 0.42026346921920776, |
| "sampling/sampling_logp_difference/max": 0.38297722339630125, |
| "sampling/sampling_logp_difference/mean": 0.003727629710920155, |
| "step": 1850, |
| "step_time": 9.405072552943603 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 927.7, |
| "completions/max_terminated_length": 927.7, |
| "completions/mean_length": 411.296875, |
| "completions/mean_terminated_length": 411.296875, |
| "completions/min_length": 250.9, |
| "completions/min_terminated_length": 250.9, |
| "entropy": 0.12066797530278564, |
| "epoch": 1.9914346895074946, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.30078125, |
| "learning_rate": 9.814100000000001e-06, |
| "loss": -0.0005, |
| "num_tokens": 105866027.0, |
| "reward": 1.040625023841858, |
| "reward_std": 0.17387359738349914, |
| "rewards/reward_accuracy/mean": 0.940625, |
| "rewards/reward_accuracy/std": 0.17387360483407974, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7835765600204467, |
| "sampling/importance_sampling_ratio/mean": 0.9941237330436706, |
| "sampling/importance_sampling_ratio/min": 0.46940397918224336, |
| "sampling/sampling_logp_difference/max": 0.2918365001678467, |
| "sampling/sampling_logp_difference/mean": 0.0038419438758865, |
| "step": 1860, |
| "step_time": 10.677912968862803 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.1143436242709867e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.1143436242709867e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 733.0, |
| "completions/max_terminated_length": 733.0, |
| "completions/mean_length": 418.5546875, |
| "completions/mean_terminated_length": 418.5546875, |
| "completions/min_length": 251.8, |
| "completions/min_terminated_length": 251.8, |
| "entropy": 0.10916378246620298, |
| "epoch": 2.0021413276231264, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.359375, |
| "learning_rate": 9.813100000000001e-06, |
| "loss": -0.0041, |
| "num_tokens": 106260718.0, |
| "reward": 1.040625023841858, |
| "reward_std": 0.2029147893190384, |
| "rewards/reward_accuracy/mean": 0.940625, |
| "rewards/reward_accuracy/std": 0.2029147908091545, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.052096796035767, |
| "sampling/importance_sampling_ratio/mean": 1.0266874134540558, |
| "sampling/importance_sampling_ratio/min": 0.4169542834162712, |
| "sampling/sampling_logp_difference/max": 0.4031745672225952, |
| "sampling/sampling_logp_difference/mean": 0.0036836348939687014, |
| "step": 1870, |
| "step_time": 8.917017789650709 |
| }, |
| { |
| "clip_ratio/high_max": 1.5903307939879596e-05, |
| "clip_ratio/high_mean": 3.975826984969899e-06, |
| "clip_ratio/low_mean": 2.7412281269789675e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.7170551119488666e-06, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 1044.3, |
| "completions/max_terminated_length": 861.6, |
| "completions/mean_length": 425.134375, |
| "completions/mean_terminated_length": 421.1573425292969, |
| "completions/min_length": 266.0, |
| "completions/min_terminated_length": 266.0, |
| "entropy": 0.1089480958878994, |
| "epoch": 2.012847965738758, |
| "frac_reward_zero_std": 0.8, |
| "grad_norm": 0.62109375, |
| "learning_rate": 9.8121e-06, |
| "loss": -0.0035, |
| "num_tokens": 106664004.0, |
| "reward": 1.0592187762260437, |
| "reward_std": 0.1847193494439125, |
| "rewards/reward_accuracy/mean": 0.959375, |
| "rewards/reward_accuracy/std": 0.18346935361623765, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 2.0842920422554014, |
| "sampling/importance_sampling_ratio/mean": 1.0031748831272125, |
| "sampling/importance_sampling_ratio/min": 0.4082653775811195, |
| "sampling/sampling_logp_difference/max": 0.4083940625190735, |
| "sampling/sampling_logp_difference/mean": 0.0036609418923035262, |
| "step": 1880, |
| "step_time": 11.837867666082456 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 580.1, |
| "completions/max_terminated_length": 580.1, |
| "completions/mean_length": 397.2046875, |
| "completions/mean_terminated_length": 397.2046875, |
| "completions/min_length": 270.9, |
| "completions/min_terminated_length": 270.9, |
| "entropy": 0.10247172918170691, |
| "epoch": 2.02355460385439, |
| "frac_reward_zero_std": 0.875, |
| "grad_norm": 0.0, |
| "learning_rate": 9.811100000000002e-06, |
| "loss": 0.0014, |
| "num_tokens": 107045415.0, |
| "reward": 1.060937523841858, |
| "reward_std": 0.15035490393638612, |
| "rewards/reward_accuracy/mean": 0.9609375, |
| "rewards/reward_accuracy/std": 0.15035490989685057, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0315518975257874, |
| "sampling/importance_sampling_ratio/mean": 0.9915335178375244, |
| "sampling/importance_sampling_ratio/min": 0.4330960988998413, |
| "sampling/sampling_logp_difference/max": 0.32977692484855653, |
| "sampling/sampling_logp_difference/mean": 0.003522640885785222, |
| "step": 1890, |
| "step_time": 7.479427301790565 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 670.3, |
| "completions/max_terminated_length": 670.3, |
| "completions/mean_length": 385.45625, |
| "completions/mean_terminated_length": 385.45625, |
| "completions/min_length": 236.9, |
| "completions/min_terminated_length": 236.9, |
| "entropy": 0.10733316815458238, |
| "epoch": 2.0342612419700212, |
| "frac_reward_zero_std": 0.8375, |
| "grad_norm": 0.318359375, |
| "learning_rate": 9.810100000000001e-06, |
| "loss": -0.0033, |
| "num_tokens": 107419739.0, |
| "reward": 1.021875023841858, |
| "reward_std": 0.23765696585178375, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.23765696883201598, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.963684117794037, |
| "sampling/importance_sampling_ratio/mean": 1.0120892226696014, |
| "sampling/importance_sampling_ratio/min": 0.38956857174634935, |
| "sampling/sampling_logp_difference/max": 0.4397976458072662, |
| "sampling/sampling_logp_difference/mean": 0.003635748312808573, |
| "step": 1900, |
| "step_time": 8.396151277469471 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 879.9, |
| "completions/max_terminated_length": 879.9, |
| "completions/mean_length": 415.5359375, |
| "completions/mean_terminated_length": 415.5359375, |
| "completions/min_length": 272.0, |
| "completions/min_terminated_length": 272.0, |
| "entropy": 0.11609016563743353, |
| "epoch": 2.044967880085653, |
| "frac_reward_zero_std": 0.75, |
| "grad_norm": 0.55859375, |
| "learning_rate": 9.8091e-06, |
| "loss": -0.0002, |
| "num_tokens": 107816914.0, |
| "reward": 0.9812500238418579, |
| "reward_std": 0.2918085515499115, |
| "rewards/reward_accuracy/mean": 0.88125, |
| "rewards/reward_accuracy/std": 0.2918085515499115, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.068711984157562, |
| "sampling/importance_sampling_ratio/mean": 0.987515902519226, |
| "sampling/importance_sampling_ratio/min": 0.3487475156784058, |
| "sampling/sampling_logp_difference/max": 0.3558837115764618, |
| "sampling/sampling_logp_difference/mean": 0.004006699123419821, |
| "step": 1910, |
| "step_time": 10.306773517513648 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.1846095731016248e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.1846095731016248e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1140.4, |
| "completions/max_terminated_length": 1140.4, |
| "completions/mean_length": 398.5515625, |
| "completions/mean_terminated_length": 398.5515625, |
| "completions/min_length": 228.2, |
| "completions/min_terminated_length": 228.2, |
| "entropy": 0.09697373434901238, |
| "epoch": 2.0556745182012848, |
| "frac_reward_zero_std": 0.825, |
| "grad_norm": 0.0, |
| "learning_rate": 9.8081e-06, |
| "loss": 0.0021, |
| "num_tokens": 108197059.0, |
| "reward": 1.037500023841858, |
| "reward_std": 0.21010553240776061, |
| "rewards/reward_accuracy/mean": 0.9375, |
| "rewards/reward_accuracy/std": 0.21010554283857347, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9592666387557984, |
| "sampling/importance_sampling_ratio/mean": 0.9950353682041169, |
| "sampling/importance_sampling_ratio/min": 0.3938357770442963, |
| "sampling/sampling_logp_difference/max": 0.3740521490573883, |
| "sampling/sampling_logp_difference/mean": 0.0036501064663752914, |
| "step": 1920, |
| "step_time": 12.471255863411352 |
| }, |
| { |
| "clip_ratio/high_max": 1.4534883666783571e-05, |
| "clip_ratio/high_mean": 3.6337209166958927e-06, |
| "clip_ratio/low_mean": 2.800179208861664e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.433900125557557e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 738.4, |
| "completions/max_terminated_length": 738.4, |
| "completions/mean_length": 402.6640625, |
| "completions/mean_terminated_length": 402.6640625, |
| "completions/min_length": 262.1, |
| "completions/min_terminated_length": 262.1, |
| "entropy": 0.09490237946156413, |
| "epoch": 2.0663811563169165, |
| "frac_reward_zero_std": 0.75, |
| "grad_norm": 0.458984375, |
| "learning_rate": 9.8071e-06, |
| "loss": 0.0035, |
| "num_tokens": 108585228.0, |
| "reward": 1.0389843940734864, |
| "reward_std": 0.2176605761051178, |
| "rewards/reward_accuracy/mean": 0.9390625, |
| "rewards/reward_accuracy/std": 0.2173842802643776, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006250000558793544, |
| "sampling/importance_sampling_ratio/max": 1.9192985415458679, |
| "sampling/importance_sampling_ratio/mean": 0.9752121865749359, |
| "sampling/importance_sampling_ratio/min": 0.3271015495061874, |
| "sampling/sampling_logp_difference/max": 0.38975948095321655, |
| "sampling/sampling_logp_difference/mean": 0.003704390209168196, |
| "step": 1930, |
| "step_time": 8.852115076733753 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.6939655072055756e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.6939655072055756e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1118.3, |
| "completions/max_terminated_length": 1118.3, |
| "completions/mean_length": 399.3, |
| "completions/mean_terminated_length": 399.3, |
| "completions/min_length": 250.5, |
| "completions/min_terminated_length": 250.5, |
| "entropy": 0.09180955342017114, |
| "epoch": 2.0770877944325483, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.16015625, |
| "learning_rate": 9.806100000000001e-06, |
| "loss": 0.0001, |
| "num_tokens": 108971260.0, |
| "reward": 1.0639843940734863, |
| "reward_std": 0.14317230880260468, |
| "rewards/reward_accuracy/mean": 0.9640625, |
| "rewards/reward_accuracy/std": 0.14292190968990326, |
| "rewards/reward_format/mean": 0.09992187693715096, |
| "rewards/reward_format/std": 0.0006250000558793544, |
| "sampling/importance_sampling_ratio/max": 1.9478360772132874, |
| "sampling/importance_sampling_ratio/mean": 0.9894565224647522, |
| "sampling/importance_sampling_ratio/min": 0.36909416019916536, |
| "sampling/sampling_logp_difference/max": 0.5193236589431762, |
| "sampling/sampling_logp_difference/mean": 0.003824052354320884, |
| "step": 1940, |
| "step_time": 12.292906090430915 |
| }, |
| { |
| "clip_ratio/high_max": 8.37801635498181e-06, |
| "clip_ratio/high_mean": 2.0945040887454525e-06, |
| "clip_ratio/low_mean": 2.325148852833081e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.419652941578533e-06, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 985.1, |
| "completions/max_terminated_length": 833.7, |
| "completions/mean_length": 425.5328125, |
| "completions/mean_terminated_length": 421.48328552246096, |
| "completions/min_length": 268.8, |
| "completions/min_terminated_length": 268.8, |
| "entropy": 0.08801116184331477, |
| "epoch": 2.08779443254818, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.435546875, |
| "learning_rate": 9.8051e-06, |
| "loss": 0.0128, |
| "num_tokens": 109372177.0, |
| "reward": 1.0232812762260437, |
| "reward_std": 0.1895739585161209, |
| "rewards/reward_accuracy/mean": 0.9234375, |
| "rewards/reward_accuracy/std": 0.18930575549602507, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 2.286674642562866, |
| "sampling/importance_sampling_ratio/mean": 1.018673402070999, |
| "sampling/importance_sampling_ratio/min": 0.3352431461215019, |
| "sampling/sampling_logp_difference/max": 0.365849769115448, |
| "sampling/sampling_logp_difference/mean": 0.003657024190761149, |
| "step": 1950, |
| "step_time": 11.224304763786495 |
| }, |
| { |
| "clip_ratio/high_max": 4.359879676485434e-05, |
| "clip_ratio/high_mean": 1.0899699191213585e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 1.0899699191213585e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 995.6, |
| "completions/max_terminated_length": 995.6, |
| "completions/mean_length": 419.775, |
| "completions/mean_terminated_length": 419.775, |
| "completions/min_length": 250.1, |
| "completions/min_terminated_length": 250.1, |
| "entropy": 0.09093467621132731, |
| "epoch": 2.0985010706638114, |
| "frac_reward_zero_std": 0.8625, |
| "grad_norm": 0.154296875, |
| "learning_rate": 9.804100000000002e-06, |
| "loss": 0.0021, |
| "num_tokens": 109773521.0, |
| "reward": 1.067187523841858, |
| "reward_std": 0.13555558323860167, |
| "rewards/reward_accuracy/mean": 0.9671875, |
| "rewards/reward_accuracy/std": 0.13555558919906616, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.3678683161735536, |
| "sampling/importance_sampling_ratio/mean": 0.9985384941101074, |
| "sampling/importance_sampling_ratio/min": 0.3270552299916744, |
| "sampling/sampling_logp_difference/max": 0.45385468006134033, |
| "sampling/sampling_logp_difference/mean": 0.003835552325472236, |
| "step": 1960, |
| "step_time": 11.26548831234686 |
| }, |
| { |
| "clip_ratio/high_max": 3.074290070799179e-05, |
| "clip_ratio/high_mean": 7.685725176997948e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 7.685725176997948e-06, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 1554.4, |
| "completions/max_terminated_length": 1263.8, |
| "completions/mean_length": 441.403125, |
| "completions/mean_terminated_length": 433.2246826171875, |
| "completions/min_length": 258.2, |
| "completions/min_terminated_length": 258.2, |
| "entropy": 0.10730418493039906, |
| "epoch": 2.109207708779443, |
| "frac_reward_zero_std": 0.6875, |
| "grad_norm": 0.443359375, |
| "learning_rate": 9.803100000000001e-06, |
| "loss": -0.0113, |
| "num_tokens": 110184867.0, |
| "reward": 1.0167969107627868, |
| "reward_std": 0.24756862223148346, |
| "rewards/reward_accuracy/mean": 0.9171875, |
| "rewards/reward_accuracy/std": 0.24628619402647017, |
| "rewards/reward_format/mean": 0.09960937723517418, |
| "rewards/reward_format/std": 0.0031250000931322573, |
| "sampling/importance_sampling_ratio/max": 2.139665973186493, |
| "sampling/importance_sampling_ratio/mean": 0.9921779274940491, |
| "sampling/importance_sampling_ratio/min": 0.27297855988144876, |
| "sampling/sampling_logp_difference/max": 0.35816174149513247, |
| "sampling/sampling_logp_difference/mean": 0.004277958930470049, |
| "step": 1970, |
| "step_time": 16.70668127029203 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.051571027550381e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.051571027550381e-06, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 2621.8, |
| "completions/max_terminated_length": 2336.5, |
| "completions/mean_length": 567.7890625, |
| "completions/mean_terminated_length": 528.207958984375, |
| "completions/min_length": 266.4, |
| "completions/min_terminated_length": 266.4, |
| "entropy": 0.1526195852085948, |
| "epoch": 2.119914346895075, |
| "frac_reward_zero_std": 0.7875, |
| "grad_norm": 0.3203125, |
| "learning_rate": 9.8021e-06, |
| "loss": -0.0045, |
| "num_tokens": 110677100.0, |
| "reward": 0.934062522649765, |
| "reward_std": 0.37235058546066285, |
| "rewards/reward_accuracy/mean": 0.8359375, |
| "rewards/reward_accuracy/std": 0.36846200823783876, |
| "rewards/reward_format/mean": 0.09812500178813935, |
| "rewards/reward_format/std": 0.01036800229921937, |
| "sampling/importance_sampling_ratio/max": 2.0899519443511965, |
| "sampling/importance_sampling_ratio/mean": 0.955902898311615, |
| "sampling/importance_sampling_ratio/min": 0.15894732400774955, |
| "sampling/sampling_logp_difference/max": 0.4875416159629822, |
| "sampling/sampling_logp_difference/mean": 0.005078140227124095, |
| "step": 1980, |
| "step_time": 28.833468145737424 |
| }, |
| { |
| "clip_ratio/high_max": 2.31481473747408e-05, |
| "clip_ratio/high_mean": 5.7870368436852e-06, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 5.7870368436852e-06, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 1582.7, |
| "completions/max_terminated_length": 1364.2, |
| "completions/mean_length": 502.4171875, |
| "completions/mean_terminated_length": 498.45342407226565, |
| "completions/min_length": 295.4, |
| "completions/min_terminated_length": 295.4, |
| "entropy": 0.11265759035013616, |
| "epoch": 2.1306209850107067, |
| "frac_reward_zero_std": 0.8125, |
| "grad_norm": 0.3125, |
| "learning_rate": 9.801100000000002e-06, |
| "loss": -0.0035, |
| "num_tokens": 111128855.0, |
| "reward": 1.0217187762260438, |
| "reward_std": 0.2245577096939087, |
| "rewards/reward_accuracy/mean": 0.921875, |
| "rewards/reward_accuracy/std": 0.2240870013833046, |
| "rewards/reward_format/mean": 0.09984375163912773, |
| "rewards/reward_format/std": 0.0012500000186264515, |
| "sampling/importance_sampling_ratio/max": 2.2145517826080323, |
| "sampling/importance_sampling_ratio/mean": 0.9961477637290954, |
| "sampling/importance_sampling_ratio/min": 0.2209709346294403, |
| "sampling/sampling_logp_difference/max": 0.466306209564209, |
| "sampling/sampling_logp_difference/mean": 0.004133241600356996, |
| "step": 1990, |
| "step_time": 16.830543696368114 |
| }, |
| { |
| "clip_ratio/high_max": 9.097525617107748e-06, |
| "clip_ratio/high_mean": 2.274381404276937e-06, |
| "clip_ratio/low_mean": 2.200704147981014e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.475085552257951e-06, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1080.8, |
| "completions/max_terminated_length": 1080.8, |
| "completions/mean_length": 487.3484375, |
| "completions/mean_terminated_length": 487.3484375, |
| "completions/min_length": 293.1, |
| "completions/min_terminated_length": 293.1, |
| "entropy": 0.10770832821726799, |
| "epoch": 2.1413276231263385, |
| "frac_reward_zero_std": 0.8875, |
| "grad_norm": 0.7890625, |
| "learning_rate": 9.800100000000001e-06, |
| "loss": 0.0038, |
| "num_tokens": 111567302.0, |
| "reward": 1.053125023841858, |
| "reward_std": 0.14287035465240477, |
| "rewards/reward_accuracy/mean": 0.953125, |
| "rewards/reward_accuracy/std": 0.14287035614252092, |
| "rewards/reward_format/mean": 0.10000000149011612, |
| "rewards/reward_format/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.2943642020225523, |
| "sampling/importance_sampling_ratio/mean": 1.0328330397605896, |
| "sampling/importance_sampling_ratio/min": 0.301421132683754, |
| "sampling/sampling_logp_difference/max": 0.39536782503128054, |
| "sampling/sampling_logp_difference/mean": 0.003932665218599141, |
| "step": 2000, |
| "step_time": 12.038470902713016 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 100000, |
| "num_input_tokens_seen": 111567302, |
| "num_train_epochs": 108, |
| "save_steps": 100, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|