| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5208333333333334, |
| "eval_steps": 500, |
| "global_step": 200, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.15000000596046448, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2029.0, |
| "completions/mean_length": 652.7667236328125, |
| "completions/mean_terminated_length": 406.5490417480469, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.3459247102340062, |
| "epoch": 0.0026041666666666665, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009550241753458977, |
| "learning_rate": 0.0, |
| "loss": 0.0013069622218608856, |
| "num_tokens": 44636.0, |
| "reward": 0.01459961012005806, |
| "reward_std": 0.6413673162460327, |
| "rewards/correctness/mean": 0.3333333432674408, |
| "rewards/correctness/std": 0.4753826856613159, |
| "rewards/length_penalty/mean": -0.31873372197151184, |
| "rewards/length_penalty/std": 0.32635030150413513, |
| "sampling/importance_sampling_ratio/max": 1.6358195543289185, |
| "sampling/importance_sampling_ratio/mean": 0.9876935482025146, |
| "sampling/importance_sampling_ratio/min": 0.6140032410621643, |
| "sampling/sampling_logp_difference/max": 0.49214398860931396, |
| "sampling/sampling_logp_difference/mean": 0.0208564605563879, |
| "step": 1, |
| "step_time": 26.742904979968444 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1836.0, |
| "completions/mean_length": 369.1000061035156, |
| "completions/mean_terminated_length": 311.2069091796875, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.3272501329580943, |
| "epoch": 0.005208333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009315946139395237, |
| "learning_rate": 5.000000000000001e-07, |
| "loss": 0.0013181539252400398, |
| "num_tokens": 71502.0, |
| "reward": 0.48644208908081055, |
| "reward_std": 0.6016252636909485, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.18022461235523224, |
| "rewards/length_penalty/std": 0.2339743673801422, |
| "sampling/importance_sampling_ratio/max": 1.6413589715957642, |
| "sampling/importance_sampling_ratio/mean": 0.9877811670303345, |
| "sampling/importance_sampling_ratio/min": 0.47745126485824585, |
| "sampling/sampling_logp_difference/max": 0.7392932176589966, |
| "sampling/sampling_logp_difference/mean": 0.02121085673570633, |
| "step": 2, |
| "step_time": 25.081525187939405 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006315211843078335, |
| "clip_ratio/high_mean": 0.0006315211843078335, |
| "clip_ratio/low_mean": 4.220477906831851e-05, |
| "clip_ratio/low_min": 4.220477906831851e-05, |
| "clip_ratio/region_mean": 0.000673725963376152, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 919.0, |
| "completions/mean_length": 268.4000244140625, |
| "completions/mean_terminated_length": 207.03448486328125, |
| "completions/min_length": 65.0, |
| "completions/min_terminated_length": 65.0, |
| "entropy": 0.27450663099686307, |
| "epoch": 0.0078125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008179417811334133, |
| "learning_rate": 1.0000000000000002e-06, |
| "loss": 0.04328472912311554, |
| "num_tokens": 92996.0, |
| "reward": 0.7356120347976685, |
| "reward_std": 0.4478960931301117, |
| "rewards/correctness/mean": 0.8666666746139526, |
| "rewards/correctness/std": 0.34280332922935486, |
| "rewards/length_penalty/mean": -0.13105468451976776, |
| "rewards/length_penalty/std": 0.17963626980781555, |
| "sampling/importance_sampling_ratio/max": 1.5385353565216064, |
| "sampling/importance_sampling_ratio/mean": 0.9898919463157654, |
| "sampling/importance_sampling_ratio/min": 0.6488208770751953, |
| "sampling/sampling_logp_difference/max": 0.4325985908508301, |
| "sampling/sampling_logp_difference/mean": 0.018099728971719742, |
| "step": 3, |
| "step_time": 24.624736201018095 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011935062599756445, |
| "clip_ratio/high_mean": 0.0011935062599756445, |
| "clip_ratio/low_mean": 5.3197150312674545e-05, |
| "clip_ratio/low_min": 5.3197150312674545e-05, |
| "clip_ratio/region_mean": 0.001246703410288319, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1035.0, |
| "completions/max_terminated_length": 1035.0, |
| "completions/mean_length": 233.60000610351562, |
| "completions/mean_terminated_length": 233.60000610351562, |
| "completions/min_length": 57.0, |
| "completions/min_terminated_length": 57.0, |
| "entropy": 0.3561083674430847, |
| "epoch": 0.010416666666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009419563226401806, |
| "learning_rate": 1.5e-06, |
| "loss": 0.029835982248187065, |
| "num_tokens": 111582.0, |
| "reward": 0.8359375596046448, |
| "reward_std": 0.2750560939311981, |
| "rewards/correctness/mean": 0.949999988079071, |
| "rewards/correctness/std": 0.2197841852903366, |
| "rewards/length_penalty/mean": -0.11406250298023224, |
| "rewards/length_penalty/std": 0.10245785117149353, |
| "sampling/importance_sampling_ratio/max": 1.3960375785827637, |
| "sampling/importance_sampling_ratio/mean": 0.9872499108314514, |
| "sampling/importance_sampling_ratio/min": 0.6510342955589294, |
| "sampling/sampling_logp_difference/max": 0.42919301986694336, |
| "sampling/sampling_logp_difference/mean": 0.021795904263854027, |
| "step": 4, |
| "step_time": 13.372587356017902 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013228575019941975, |
| "clip_ratio/high_mean": 0.0013228575019941975, |
| "clip_ratio/low_mean": 0.0002650423363472025, |
| "clip_ratio/low_min": 0.0002650423363472025, |
| "clip_ratio/region_mean": 0.0015878998383414, |
| "completions/clipped_ratio": 0.0833333358168602, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1780.0, |
| "completions/mean_length": 439.8333435058594, |
| "completions/mean_terminated_length": 293.6363525390625, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.35197166601816815, |
| "epoch": 0.013020833333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01132192276418209, |
| "learning_rate": 2.0000000000000003e-06, |
| "loss": 0.04111845791339874, |
| "num_tokens": 142332.0, |
| "reward": 0.40190431475639343, |
| "reward_std": 0.6367157697677612, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.2147623747587204, |
| "rewards/length_penalty/std": 0.2856442928314209, |
| "sampling/importance_sampling_ratio/max": 1.559218168258667, |
| "sampling/importance_sampling_ratio/mean": 0.9879145622253418, |
| "sampling/importance_sampling_ratio/min": 0.523356556892395, |
| "sampling/sampling_logp_difference/max": 0.6474922895431519, |
| "sampling/sampling_logp_difference/mean": 0.020651621744036674, |
| "step": 5, |
| "step_time": 25.51464084163308 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013545737989867728, |
| "clip_ratio/high_mean": 0.0013545737989867728, |
| "clip_ratio/low_mean": 0.0002838041521802855, |
| "clip_ratio/low_min": 0.0002838041521802855, |
| "clip_ratio/region_mean": 0.001638377948741739, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1782.0, |
| "completions/mean_length": 495.61669921875, |
| "completions/mean_terminated_length": 413.91229248046875, |
| "completions/min_length": 60.0, |
| "completions/min_terminated_length": 60.0, |
| "entropy": 0.42050177852312726, |
| "epoch": 0.015625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0115534458309412, |
| "learning_rate": 2.5e-06, |
| "loss": 0.040411680936813354, |
| "num_tokens": 177289.0, |
| "reward": 0.09133300930261612, |
| "reward_std": 0.6072342395782471, |
| "rewards/correctness/mean": 0.3333333432674408, |
| "rewards/correctness/std": 0.4753826856613159, |
| "rewards/length_penalty/mean": -0.24200032651424408, |
| "rewards/length_penalty/std": 0.2640933394432068, |
| "sampling/importance_sampling_ratio/max": 1.810820460319519, |
| "sampling/importance_sampling_ratio/mean": 0.984971821308136, |
| "sampling/importance_sampling_ratio/min": 0.6473142504692078, |
| "sampling/sampling_logp_difference/max": 0.5937800407409668, |
| "sampling/sampling_logp_difference/mean": 0.024809231981635094, |
| "step": 6, |
| "step_time": 25.44005388393998 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007666738335198412, |
| "clip_ratio/high_mean": 0.0007666738335198412, |
| "clip_ratio/low_mean": 0.00010288066308324535, |
| "clip_ratio/low_min": 0.00010288066308324535, |
| "clip_ratio/region_mean": 0.0008695544966030866, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1165.0, |
| "completions/max_terminated_length": 1165.0, |
| "completions/mean_length": 198.45001220703125, |
| "completions/mean_terminated_length": 198.45001220703125, |
| "completions/min_length": 51.0, |
| "completions/min_terminated_length": 51.0, |
| "entropy": 0.22184078147013983, |
| "epoch": 0.018229166666666668, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007696207612752914, |
| "learning_rate": 3e-06, |
| "loss": -0.0025737816467881203, |
| "num_tokens": 192996.0, |
| "reward": 0.3031006157398224, |
| "reward_std": 0.49256783723831177, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49403220415115356, |
| "rewards/length_penalty/mean": -0.09689941257238388, |
| "rewards/length_penalty/std": 0.07940318435430527, |
| "sampling/importance_sampling_ratio/max": 1.3774839639663696, |
| "sampling/importance_sampling_ratio/mean": 0.9918273091316223, |
| "sampling/importance_sampling_ratio/min": 0.5823128819465637, |
| "sampling/sampling_logp_difference/max": 0.5407474040985107, |
| "sampling/sampling_logp_difference/mean": 0.015205474570393562, |
| "step": 7, |
| "step_time": 13.684738623909652 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005758648864381636, |
| "clip_ratio/high_mean": 0.0005758648864381636, |
| "clip_ratio/low_mean": 6.324946540795888e-05, |
| "clip_ratio/low_min": 6.324946540795888e-05, |
| "clip_ratio/region_mean": 0.0006391143591220801, |
| "completions/clipped_ratio": 0.11666667461395264, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1549.0, |
| "completions/mean_length": 469.2333679199219, |
| "completions/mean_terminated_length": 260.71697998046875, |
| "completions/min_length": 51.0, |
| "completions/min_terminated_length": 51.0, |
| "entropy": 0.3752252707878749, |
| "epoch": 0.020833333333333332, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010931719094514847, |
| "learning_rate": 3.5e-06, |
| "loss": 7.687229663133621e-05, |
| "num_tokens": 225210.0, |
| "reward": 0.3208821713924408, |
| "reward_std": 0.6954562664031982, |
| "rewards/correctness/mean": 0.550000011920929, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.22911784052848816, |
| "rewards/length_penalty/std": 0.3020681142807007, |
| "sampling/importance_sampling_ratio/max": 1.3793786764144897, |
| "sampling/importance_sampling_ratio/mean": 0.9861665964126587, |
| "sampling/importance_sampling_ratio/min": 0.5893210172653198, |
| "sampling/sampling_logp_difference/max": 0.5287842750549316, |
| "sampling/sampling_logp_difference/mean": 0.022938480600714684, |
| "step": 8, |
| "step_time": 26.051152772968635 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013098372340512772, |
| "clip_ratio/high_mean": 0.0013098372340512772, |
| "clip_ratio/low_mean": 9.053051083659132e-05, |
| "clip_ratio/low_min": 9.053051083659132e-05, |
| "clip_ratio/region_mean": 0.0014003677448878686, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 440.0, |
| "completions/max_terminated_length": 440.0, |
| "completions/mean_length": 176.4166717529297, |
| "completions/mean_terminated_length": 176.4166717529297, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.2628745312492053, |
| "epoch": 0.0234375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00750648882240057, |
| "learning_rate": 4.000000000000001e-06, |
| "loss": -0.00012696627527475357, |
| "num_tokens": 239705.0, |
| "reward": 0.6305257678031921, |
| "reward_std": 0.46693822741508484, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.45441964268684387, |
| "rewards/length_penalty/mean": -0.0861409530043602, |
| "rewards/length_penalty/std": 0.03498101606965065, |
| "sampling/importance_sampling_ratio/max": 1.459780216217041, |
| "sampling/importance_sampling_ratio/mean": 0.9906548857688904, |
| "sampling/importance_sampling_ratio/min": 0.5087599754333496, |
| "sampling/sampling_logp_difference/max": 0.6757789850234985, |
| "sampling/sampling_logp_difference/mean": 0.017864875495433807, |
| "step": 9, |
| "step_time": 6.08875353098847 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011549753883931164, |
| "clip_ratio/high_mean": 0.0011549753883931164, |
| "clip_ratio/low_mean": 9.639483566085498e-05, |
| "clip_ratio/low_min": 9.639483566085498e-05, |
| "clip_ratio/region_mean": 0.0012513702240539715, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1169.0, |
| "completions/mean_length": 302.933349609375, |
| "completions/mean_terminated_length": 273.3559265136719, |
| "completions/min_length": 92.0, |
| "completions/min_terminated_length": 92.0, |
| "entropy": 0.27069761604070663, |
| "epoch": 0.026041666666666668, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008138449862599373, |
| "learning_rate": 4.5e-06, |
| "loss": 0.013499054126441479, |
| "num_tokens": 264061.0, |
| "reward": 0.4520833492279053, |
| "reward_std": 0.5371150374412537, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.14791665971279144, |
| "rewards/length_penalty/std": 0.15711446106433868, |
| "sampling/importance_sampling_ratio/max": 1.5805463790893555, |
| "sampling/importance_sampling_ratio/mean": 0.9906094074249268, |
| "sampling/importance_sampling_ratio/min": 0.44420018792152405, |
| "sampling/sampling_logp_difference/max": 0.8114800453186035, |
| "sampling/sampling_logp_difference/mean": 0.017728246748447418, |
| "step": 10, |
| "step_time": 24.401646023849025 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007490230442878479, |
| "clip_ratio/high_mean": 0.0007490230442878479, |
| "clip_ratio/low_mean": 0.0002648754937884708, |
| "clip_ratio/low_min": 0.0002648754937884708, |
| "clip_ratio/region_mean": 0.0010138985235244036, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1843.0, |
| "completions/mean_length": 481.66668701171875, |
| "completions/mean_terminated_length": 307.629638671875, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.292586291829745, |
| "epoch": 0.028645833333333332, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006925498601049185, |
| "learning_rate": 5e-06, |
| "loss": -0.0027555525302886963, |
| "num_tokens": 297561.0, |
| "reward": 0.24814453721046448, |
| "reward_std": 0.6807762384414673, |
| "rewards/correctness/mean": 0.4833333194255829, |
| "rewards/correctness/std": 0.5039392709732056, |
| "rewards/length_penalty/mean": -0.2351887971162796, |
| "rewards/length_penalty/std": 0.31384536623954773, |
| "sampling/importance_sampling_ratio/max": 1.4562650918960571, |
| "sampling/importance_sampling_ratio/mean": 0.9886733293533325, |
| "sampling/importance_sampling_ratio/min": 0.6622587442398071, |
| "sampling/sampling_logp_difference/max": 0.41209888458251953, |
| "sampling/sampling_logp_difference/mean": 0.018755722790956497, |
| "step": 11, |
| "step_time": 25.377477446803823 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011631436694491033, |
| "clip_ratio/high_mean": 0.0011631436694491033, |
| "clip_ratio/low_mean": 0.00026016056168979657, |
| "clip_ratio/low_min": 0.00026016056168979657, |
| "clip_ratio/region_mean": 0.0014233042408401768, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1636.0, |
| "completions/max_terminated_length": 1636.0, |
| "completions/mean_length": 299.63336181640625, |
| "completions/mean_terminated_length": 299.63336181640625, |
| "completions/min_length": 100.0, |
| "completions/min_terminated_length": 100.0, |
| "entropy": 0.3268725921710332, |
| "epoch": 0.03125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007576283533126116, |
| "learning_rate": 5.500000000000001e-06, |
| "loss": 0.011999893002212048, |
| "num_tokens": 320989.0, |
| "reward": 0.487028032541275, |
| "reward_std": 0.5599981546401978, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.14630533754825592, |
| "rewards/length_penalty/std": 0.14096499979496002, |
| "sampling/importance_sampling_ratio/max": 1.4403142929077148, |
| "sampling/importance_sampling_ratio/mean": 0.9886054396629333, |
| "sampling/importance_sampling_ratio/min": 0.6616308093070984, |
| "sampling/sampling_logp_difference/max": 0.41304755210876465, |
| "sampling/sampling_logp_difference/mean": 0.020671943202614784, |
| "step": 12, |
| "step_time": 20.598799626808614 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008587691312034925, |
| "clip_ratio/high_mean": 0.0008587691312034925, |
| "clip_ratio/low_mean": 0.0001680954786327978, |
| "clip_ratio/low_min": 0.0001680954786327978, |
| "clip_ratio/region_mean": 0.0010268646001350135, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 469.0, |
| "completions/max_terminated_length": 469.0, |
| "completions/mean_length": 196.11668395996094, |
| "completions/mean_terminated_length": 196.11668395996094, |
| "completions/min_length": 45.0, |
| "completions/min_terminated_length": 45.0, |
| "entropy": 0.243132712940375, |
| "epoch": 0.033854166666666664, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0058483704924583435, |
| "learning_rate": 6e-06, |
| "loss": 0.0022401618771255016, |
| "num_tokens": 337526.0, |
| "reward": 0.5875732898712158, |
| "reward_std": 0.4762917160987854, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.09576009213924408, |
| "rewards/length_penalty/std": 0.04025096073746681, |
| "sampling/importance_sampling_ratio/max": 1.3507492542266846, |
| "sampling/importance_sampling_ratio/mean": 0.9909545183181763, |
| "sampling/importance_sampling_ratio/min": 0.6604217290878296, |
| "sampling/sampling_logp_difference/max": 0.41487669944763184, |
| "sampling/sampling_logp_difference/mean": 0.016091017052531242, |
| "step": 13, |
| "step_time": 6.548263439908624 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013289929241485272, |
| "clip_ratio/high_mean": 0.0013289929241485272, |
| "clip_ratio/low_mean": 0.00018298412639220865, |
| "clip_ratio/low_min": 0.00018298412639220865, |
| "clip_ratio/region_mean": 0.0015119770153736074, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1901.0, |
| "completions/mean_length": 573.3833618164062, |
| "completions/mean_terminated_length": 548.3898315429688, |
| "completions/min_length": 66.0, |
| "completions/min_terminated_length": 66.0, |
| "entropy": 0.4546853502591451, |
| "epoch": 0.036458333333333336, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012831938453018665, |
| "learning_rate": 6.5000000000000004e-06, |
| "loss": -0.016117742285132408, |
| "num_tokens": 378939.0, |
| "reward": 0.3366943597793579, |
| "reward_std": 0.5785391330718994, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.2799723446369171, |
| "rewards/length_penalty/std": 0.2935960292816162, |
| "sampling/importance_sampling_ratio/max": 1.4548825025558472, |
| "sampling/importance_sampling_ratio/mean": 0.9844425320625305, |
| "sampling/importance_sampling_ratio/min": 0.43611282110214233, |
| "sampling/sampling_logp_difference/max": 0.8298543691635132, |
| "sampling/sampling_logp_difference/mean": 0.025902364403009415, |
| "step": 14, |
| "step_time": 26.968123928876594 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008295439038192853, |
| "clip_ratio/high_mean": 0.0008295439038192853, |
| "clip_ratio/low_mean": 8.032128486471872e-05, |
| "clip_ratio/low_min": 8.032128486471872e-05, |
| "clip_ratio/region_mean": 0.000909865188684004, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 772.0, |
| "completions/max_terminated_length": 772.0, |
| "completions/mean_length": 208.06668090820312, |
| "completions/mean_terminated_length": 208.06668090820312, |
| "completions/min_length": 59.0, |
| "completions/min_terminated_length": 59.0, |
| "entropy": 0.25799742837746936, |
| "epoch": 0.0390625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006846250966191292, |
| "learning_rate": 7e-06, |
| "loss": 0.0016769105568528175, |
| "num_tokens": 395493.0, |
| "reward": 0.4650716483592987, |
| "reward_std": 0.5400400161743164, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.10159505158662796, |
| "rewards/length_penalty/std": 0.05763912945985794, |
| "sampling/importance_sampling_ratio/max": 1.3983293771743774, |
| "sampling/importance_sampling_ratio/mean": 0.9911385774612427, |
| "sampling/importance_sampling_ratio/min": 0.6694991588592529, |
| "sampling/sampling_logp_difference/max": 0.40122532844543457, |
| "sampling/sampling_logp_difference/mean": 0.01631416380405426, |
| "step": 15, |
| "step_time": 9.458079774631187 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012162097846157849, |
| "clip_ratio/high_mean": 0.0012162097846157849, |
| "clip_ratio/low_mean": 6.177415586231898e-05, |
| "clip_ratio/low_min": 6.177415586231898e-05, |
| "clip_ratio/region_mean": 0.0012779839356274654, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1419.0, |
| "completions/max_terminated_length": 1419.0, |
| "completions/mean_length": 237.78334045410156, |
| "completions/mean_terminated_length": 237.78334045410156, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.2349324276049932, |
| "epoch": 0.041666666666666664, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007547018583863974, |
| "learning_rate": 7.500000000000001e-06, |
| "loss": 0.003749743103981018, |
| "num_tokens": 414730.0, |
| "reward": 0.6172282099723816, |
| "reward_std": 0.4614962041378021, |
| "rewards/correctness/mean": 0.7333333492279053, |
| "rewards/correctness/std": 0.4459484815597534, |
| "rewards/length_penalty/mean": -0.11610514670610428, |
| "rewards/length_penalty/std": 0.10832378268241882, |
| "sampling/importance_sampling_ratio/max": 1.4435471296310425, |
| "sampling/importance_sampling_ratio/mean": 0.9914823174476624, |
| "sampling/importance_sampling_ratio/min": 0.6476690173149109, |
| "sampling/sampling_logp_difference/max": 0.434375524520874, |
| "sampling/sampling_logp_difference/mean": 0.01567167416214943, |
| "step": 16, |
| "step_time": 16.84836908383295 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010969881259370595, |
| "clip_ratio/high_mean": 0.0010969881259370595, |
| "clip_ratio/low_mean": 0.00037887512977855903, |
| "clip_ratio/low_min": 0.00037887512977855903, |
| "clip_ratio/region_mean": 0.0014758632460143417, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1508.0, |
| "completions/mean_length": 472.10003662109375, |
| "completions/mean_terminated_length": 417.75860595703125, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.32917838792006177, |
| "epoch": 0.044270833333333336, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011362938210368156, |
| "learning_rate": 8.000000000000001e-06, |
| "loss": 0.03082355484366417, |
| "num_tokens": 447696.0, |
| "reward": 0.43614912033081055, |
| "reward_std": 0.5708247423171997, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.23051758110523224, |
| "rewards/length_penalty/std": 0.23263540863990784, |
| "sampling/importance_sampling_ratio/max": 1.377583622932434, |
| "sampling/importance_sampling_ratio/mean": 0.9881200790405273, |
| "sampling/importance_sampling_ratio/min": 0.5658073425292969, |
| "sampling/sampling_logp_difference/max": 0.5695016384124756, |
| "sampling/sampling_logp_difference/mean": 0.020033184438943863, |
| "step": 17, |
| "step_time": 26.912937578046694 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009323107903279985, |
| "clip_ratio/high_mean": 0.0009323107903279985, |
| "clip_ratio/low_mean": 5.49510926551496e-05, |
| "clip_ratio/low_min": 5.49510926551496e-05, |
| "clip_ratio/region_mean": 0.000987261882983148, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 935.0, |
| "completions/mean_length": 278.1500244140625, |
| "completions/mean_terminated_length": 248.1525421142578, |
| "completions/min_length": 55.0, |
| "completions/min_terminated_length": 55.0, |
| "entropy": 0.31834355493386585, |
| "epoch": 0.046875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008774375542998314, |
| "learning_rate": 8.5e-06, |
| "loss": 0.011353441514074802, |
| "num_tokens": 467755.0, |
| "reward": 0.6308512687683105, |
| "reward_std": 0.48353394865989685, |
| "rewards/correctness/mean": 0.7666666507720947, |
| "rewards/correctness/std": 0.42652183771133423, |
| "rewards/length_penalty/mean": -0.13581542670726776, |
| "rewards/length_penalty/std": 0.14672990143299103, |
| "sampling/importance_sampling_ratio/max": 1.517329454421997, |
| "sampling/importance_sampling_ratio/mean": 0.9887821674346924, |
| "sampling/importance_sampling_ratio/min": 0.6535218954086304, |
| "sampling/sampling_logp_difference/max": 0.42537927627563477, |
| "sampling/sampling_logp_difference/mean": 0.02034401334822178, |
| "step": 18, |
| "step_time": 23.62653182190843 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007502625812776387, |
| "clip_ratio/high_mean": 0.0007502625812776387, |
| "clip_ratio/low_mean": 9.759989673815046e-05, |
| "clip_ratio/low_min": 9.759989673815046e-05, |
| "clip_ratio/region_mean": 0.0008478624828664275, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1781.0, |
| "completions/mean_length": 366.36669921875, |
| "completions/mean_terminated_length": 337.8644104003906, |
| "completions/min_length": 49.0, |
| "completions/min_terminated_length": 49.0, |
| "entropy": 0.2429429367184639, |
| "epoch": 0.049479166666666664, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010713494382798672, |
| "learning_rate": 9e-06, |
| "loss": 0.020396927371621132, |
| "num_tokens": 493827.0, |
| "reward": 0.3211100399494171, |
| "reward_std": 0.6473668217658997, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5042194724082947, |
| "rewards/length_penalty/mean": -0.17888997495174408, |
| "rewards/length_penalty/std": 0.2369253784418106, |
| "sampling/importance_sampling_ratio/max": 1.4490201473236084, |
| "sampling/importance_sampling_ratio/mean": 0.991335928440094, |
| "sampling/importance_sampling_ratio/min": 0.6717814803123474, |
| "sampling/sampling_logp_difference/max": 0.39782214164733887, |
| "sampling/sampling_logp_difference/mean": 0.01569700427353382, |
| "step": 19, |
| "step_time": 24.195959369419143 |
| }, |
| { |
| "clip_ratio/high_max": 0.000911957016796805, |
| "clip_ratio/high_mean": 0.000911957016796805, |
| "clip_ratio/low_mean": 0.00029282177274581045, |
| "clip_ratio/low_min": 0.00029282177274581045, |
| "clip_ratio/region_mean": 0.0012047787895426154, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 463.0, |
| "completions/max_terminated_length": 463.0, |
| "completions/mean_length": 212.2666778564453, |
| "completions/mean_terminated_length": 212.2666778564453, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.28134913245836896, |
| "epoch": 0.052083333333333336, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007756337057799101, |
| "learning_rate": 9.5e-06, |
| "loss": 0.0017761585768312216, |
| "num_tokens": 510403.0, |
| "reward": 0.47968751192092896, |
| "reward_std": 0.4936433732509613, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.10364583134651184, |
| "rewards/length_penalty/std": 0.03875266760587692, |
| "sampling/importance_sampling_ratio/max": 1.368051290512085, |
| "sampling/importance_sampling_ratio/mean": 0.9894872307777405, |
| "sampling/importance_sampling_ratio/min": 0.6419931054115295, |
| "sampling/sampling_logp_difference/max": 0.4431777000427246, |
| "sampling/sampling_logp_difference/mean": 0.018558764830231667, |
| "step": 20, |
| "step_time": 6.607078290078789 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007877835256901259, |
| "clip_ratio/high_mean": 0.0007877835256901259, |
| "clip_ratio/low_mean": 0.0002678541592710341, |
| "clip_ratio/low_min": 0.0002678541592710341, |
| "clip_ratio/region_mean": 0.0010556376946624368, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1066.0, |
| "completions/mean_length": 268.66668701171875, |
| "completions/mean_terminated_length": 238.5084686279297, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.34817997614542645, |
| "epoch": 0.0546875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007094081956893206, |
| "learning_rate": 1e-05, |
| "loss": -0.0031021148897707462, |
| "num_tokens": 532073.0, |
| "reward": 0.23548178374767303, |
| "reward_std": 0.5344337821006775, |
| "rewards/correctness/mean": 0.36666667461395264, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.1311848908662796, |
| "rewards/length_penalty/std": 0.1365545392036438, |
| "sampling/importance_sampling_ratio/max": 1.4603371620178223, |
| "sampling/importance_sampling_ratio/mean": 0.9874995946884155, |
| "sampling/importance_sampling_ratio/min": 0.6078290939331055, |
| "sampling/sampling_logp_difference/max": 0.49786150455474854, |
| "sampling/sampling_logp_difference/mean": 0.021213697269558907, |
| "step": 21, |
| "step_time": 24.62822749861516 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004280722835877289, |
| "clip_ratio/high_mean": 0.0004280722835877289, |
| "clip_ratio/low_mean": 0.00037680995107317966, |
| "clip_ratio/low_min": 0.00037680995107317966, |
| "clip_ratio/region_mean": 0.0008048822346609086, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1468.0, |
| "completions/max_terminated_length": 1468.0, |
| "completions/mean_length": 250.10000610351562, |
| "completions/mean_terminated_length": 250.10000610351562, |
| "completions/min_length": 80.0, |
| "completions/min_terminated_length": 80.0, |
| "entropy": 0.2983148694038391, |
| "epoch": 0.057291666666666664, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006588826887309551, |
| "learning_rate": 1e-05, |
| "loss": 0.02187872678041458, |
| "num_tokens": 551209.0, |
| "reward": 0.3112142086029053, |
| "reward_std": 0.5358970165252686, |
| "rewards/correctness/mean": 0.4333333373069763, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.12211914360523224, |
| "rewards/length_penalty/std": 0.10871647298336029, |
| "sampling/importance_sampling_ratio/max": 1.3194538354873657, |
| "sampling/importance_sampling_ratio/mean": 0.9891497492790222, |
| "sampling/importance_sampling_ratio/min": 0.6616003513336182, |
| "sampling/sampling_logp_difference/max": 0.41309356689453125, |
| "sampling/sampling_logp_difference/mean": 0.018763074651360512, |
| "step": 22, |
| "step_time": 18.03743294836022 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013897405976119142, |
| "clip_ratio/high_mean": 0.0013897405976119142, |
| "clip_ratio/low_mean": 3.846449787185217e-05, |
| "clip_ratio/low_min": 3.846449787185217e-05, |
| "clip_ratio/region_mean": 0.0014282050833571702, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1120.0, |
| "completions/max_terminated_length": 1120.0, |
| "completions/mean_length": 362.4666748046875, |
| "completions/mean_terminated_length": 362.4666748046875, |
| "completions/min_length": 102.0, |
| "completions/min_terminated_length": 102.0, |
| "entropy": 0.2391252319018046, |
| "epoch": 0.059895833333333336, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007778684142976999, |
| "learning_rate": 1e-05, |
| "loss": -0.007375192362815142, |
| "num_tokens": 577707.0, |
| "reward": 0.4896810054779053, |
| "reward_std": 0.4859490394592285, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753826856613159, |
| "rewards/length_penalty/mean": -0.17698568105697632, |
| "rewards/length_penalty/std": 0.11819574236869812, |
| "sampling/importance_sampling_ratio/max": 1.3959320783615112, |
| "sampling/importance_sampling_ratio/mean": 0.9913745522499084, |
| "sampling/importance_sampling_ratio/min": 0.5769467949867249, |
| "sampling/sampling_logp_difference/max": 0.5500051975250244, |
| "sampling/sampling_logp_difference/mean": 0.015296061523258686, |
| "step": 23, |
| "step_time": 14.198447100119665 |
| }, |
| { |
| "clip_ratio/high_max": 0.0017589333777626355, |
| "clip_ratio/high_mean": 0.0017589333777626355, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0017589333777626355, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 950.0, |
| "completions/max_terminated_length": 950.0, |
| "completions/mean_length": 191.83334350585938, |
| "completions/mean_terminated_length": 191.83334350585938, |
| "completions/min_length": 63.0, |
| "completions/min_terminated_length": 63.0, |
| "entropy": 0.3097405756513278, |
| "epoch": 0.0625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006801938638091087, |
| "learning_rate": 1e-05, |
| "loss": 0.02186577394604683, |
| "num_tokens": 592757.0, |
| "reward": 0.6396647691726685, |
| "reward_std": 0.4688263535499573, |
| "rewards/correctness/mean": 0.7333333492279053, |
| "rewards/correctness/std": 0.4459484815597534, |
| "rewards/length_penalty/mean": -0.0936686173081398, |
| "rewards/length_penalty/std": 0.07443346083164215, |
| "sampling/importance_sampling_ratio/max": 1.4199635982513428, |
| "sampling/importance_sampling_ratio/mean": 0.9886044859886169, |
| "sampling/importance_sampling_ratio/min": 0.6691972613334656, |
| "sampling/sampling_logp_difference/max": 0.4016764163970947, |
| "sampling/sampling_logp_difference/mean": 0.019662244245409966, |
| "step": 24, |
| "step_time": 11.236951956292614 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011906841342958312, |
| "clip_ratio/high_mean": 0.0011906841342958312, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011906841342958312, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 513.0, |
| "completions/max_terminated_length": 513.0, |
| "completions/mean_length": 217.10000610351562, |
| "completions/mean_terminated_length": 217.10000610351562, |
| "completions/min_length": 65.0, |
| "completions/min_terminated_length": 65.0, |
| "entropy": 0.28966160118579865, |
| "epoch": 0.06510416666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0060674045234918594, |
| "learning_rate": 1e-05, |
| "loss": -0.004418151453137398, |
| "num_tokens": 610783.0, |
| "reward": 0.2773274779319763, |
| "reward_std": 0.5059720277786255, |
| "rewards/correctness/mean": 0.38333332538604736, |
| "rewards/correctness/std": 0.4903014302253723, |
| "rewards/length_penalty/mean": -0.10600586235523224, |
| "rewards/length_penalty/std": 0.04863846302032471, |
| "sampling/importance_sampling_ratio/max": 1.395207405090332, |
| "sampling/importance_sampling_ratio/mean": 0.9902146458625793, |
| "sampling/importance_sampling_ratio/min": 0.6575587391853333, |
| "sampling/sampling_logp_difference/max": 0.4192211627960205, |
| "sampling/sampling_logp_difference/mean": 0.018368102610111237, |
| "step": 25, |
| "step_time": 7.690814247820526 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013653153049138684, |
| "clip_ratio/high_mean": 0.0013653153049138684, |
| "clip_ratio/low_mean": 4.612333335292836e-05, |
| "clip_ratio/low_min": 4.612333335292836e-05, |
| "clip_ratio/region_mean": 0.001411438638266797, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1977.0, |
| "completions/max_terminated_length": 1977.0, |
| "completions/mean_length": 557.7333374023438, |
| "completions/mean_terminated_length": 557.7333374023438, |
| "completions/min_length": 141.0, |
| "completions/min_terminated_length": 141.0, |
| "entropy": 0.3817978302637736, |
| "epoch": 0.06770833333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011830233037471771, |
| "learning_rate": 1e-05, |
| "loss": 0.017222050577402115, |
| "num_tokens": 648717.0, |
| "reward": 0.2776692807674408, |
| "reward_std": 0.5990990996360779, |
| "rewards/correctness/mean": 0.550000011920929, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.27233073115348816, |
| "rewards/length_penalty/std": 0.20231440663337708, |
| "sampling/importance_sampling_ratio/max": 1.4403636455535889, |
| "sampling/importance_sampling_ratio/mean": 0.986710786819458, |
| "sampling/importance_sampling_ratio/min": 0.530950129032135, |
| "sampling/sampling_logp_difference/max": 0.633087158203125, |
| "sampling/sampling_logp_difference/mean": 0.0223115012049675, |
| "step": 26, |
| "step_time": 24.442186386790127 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011340714942586299, |
| "clip_ratio/high_mean": 0.0011340714942586299, |
| "clip_ratio/low_mean": 7.241653899351756e-05, |
| "clip_ratio/low_min": 7.241653899351756e-05, |
| "clip_ratio/region_mean": 0.0012064880332521473, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1725.0, |
| "completions/max_terminated_length": 1725.0, |
| "completions/mean_length": 315.0833435058594, |
| "completions/mean_terminated_length": 315.0833435058594, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.24167432139317194, |
| "epoch": 0.0703125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007823715917766094, |
| "learning_rate": 1e-05, |
| "loss": 0.01304696872830391, |
| "num_tokens": 671642.0, |
| "reward": 0.6461507678031921, |
| "reward_std": 0.48402687907218933, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4033755958080292, |
| "rewards/length_penalty/mean": -0.1538492888212204, |
| "rewards/length_penalty/std": 0.13724347949028015, |
| "sampling/importance_sampling_ratio/max": 1.3459274768829346, |
| "sampling/importance_sampling_ratio/mean": 0.9914031624794006, |
| "sampling/importance_sampling_ratio/min": 0.6385123133659363, |
| "sampling/sampling_logp_difference/max": 0.44861435890197754, |
| "sampling/sampling_logp_difference/mean": 0.016104314476251602, |
| "step": 27, |
| "step_time": 21.22021480393596 |
| }, |
| { |
| "clip_ratio/high_max": 0.00059348529127116, |
| "clip_ratio/high_mean": 0.00059348529127116, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00059348529127116, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 449.0, |
| "completions/max_terminated_length": 449.0, |
| "completions/mean_length": 154.58334350585938, |
| "completions/mean_terminated_length": 154.58334350585938, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.21334969500700632, |
| "epoch": 0.07291666666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005059989169239998, |
| "learning_rate": 1e-05, |
| "loss": 0.0004967651329934597, |
| "num_tokens": 685057.0, |
| "reward": 0.5745198726654053, |
| "reward_std": 0.49488767981529236, |
| "rewards/correctness/mean": 0.6499999761581421, |
| "rewards/correctness/std": 0.48099473118782043, |
| "rewards/length_penalty/mean": -0.0754801407456398, |
| "rewards/length_penalty/std": 0.03555199131369591, |
| "sampling/importance_sampling_ratio/max": 1.4033803939819336, |
| "sampling/importance_sampling_ratio/mean": 0.9926474094390869, |
| "sampling/importance_sampling_ratio/min": 0.6647396683692932, |
| "sampling/sampling_logp_difference/max": 0.4083597660064697, |
| "sampling/sampling_logp_difference/mean": 0.014959922060370445, |
| "step": 28, |
| "step_time": 6.031793837202713 |
| }, |
| { |
| "clip_ratio/high_max": 0.001032077067065984, |
| "clip_ratio/high_mean": 0.001032077067065984, |
| "clip_ratio/low_mean": 4.5687134843319654e-05, |
| "clip_ratio/low_min": 4.5687134843319654e-05, |
| "clip_ratio/region_mean": 0.0010777642019093037, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1473.0, |
| "completions/max_terminated_length": 1473.0, |
| "completions/mean_length": 268.1000061035156, |
| "completions/mean_terminated_length": 268.1000061035156, |
| "completions/min_length": 69.0, |
| "completions/min_terminated_length": 69.0, |
| "entropy": 0.3026011536518733, |
| "epoch": 0.07552083333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00811783131211996, |
| "learning_rate": 1e-05, |
| "loss": 0.01565806195139885, |
| "num_tokens": 704933.0, |
| "reward": 0.569091796875, |
| "reward_std": 0.5122504234313965, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.13090820610523224, |
| "rewards/length_penalty/std": 0.11997226625680923, |
| "sampling/importance_sampling_ratio/max": 1.470341682434082, |
| "sampling/importance_sampling_ratio/mean": 0.988998293876648, |
| "sampling/importance_sampling_ratio/min": 0.6175099015235901, |
| "sampling/sampling_logp_difference/max": 0.48206019401550293, |
| "sampling/sampling_logp_difference/mean": 0.019445886835455894, |
| "step": 29, |
| "step_time": 17.32524182787165 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015591846701378624, |
| "clip_ratio/high_mean": 0.0015591846701378624, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0015591846701378624, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 352.0, |
| "completions/max_terminated_length": 352.0, |
| "completions/mean_length": 175.20001220703125, |
| "completions/mean_terminated_length": 175.20001220703125, |
| "completions/min_length": 77.0, |
| "completions/min_terminated_length": 77.0, |
| "entropy": 0.23111048340797424, |
| "epoch": 0.078125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004279103595763445, |
| "learning_rate": 1e-05, |
| "loss": -0.0021134447306394577, |
| "num_tokens": 719125.0, |
| "reward": 0.7144531607627869, |
| "reward_std": 0.40492507815361023, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4033755958080292, |
| "rewards/length_penalty/mean": -0.08554687350988388, |
| "rewards/length_penalty/std": 0.03281773626804352, |
| "sampling/importance_sampling_ratio/max": 1.4519866704940796, |
| "sampling/importance_sampling_ratio/mean": 0.9913766980171204, |
| "sampling/importance_sampling_ratio/min": 0.5715975165367126, |
| "sampling/sampling_logp_difference/max": 0.5593202114105225, |
| "sampling/sampling_logp_difference/mean": 0.015721017494797707, |
| "step": 30, |
| "step_time": 5.281871759798378 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008947264698993725, |
| "clip_ratio/high_mean": 0.0008947264698993725, |
| "clip_ratio/low_mean": 0.00013535359418407703, |
| "clip_ratio/low_min": 0.00013535359418407703, |
| "clip_ratio/region_mean": 0.0010300800737847264, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1540.0, |
| "completions/mean_length": 381.20001220703125, |
| "completions/mean_terminated_length": 323.7241516113281, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.3615889251232147, |
| "epoch": 0.08072916666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008097507059574127, |
| "learning_rate": 1e-05, |
| "loss": 0.029858484864234924, |
| "num_tokens": 747887.0, |
| "reward": 0.5305339097976685, |
| "reward_std": 0.5316849946975708, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.45441964268684387, |
| "rewards/length_penalty/mean": -0.18613281846046448, |
| "rewards/length_penalty/std": 0.20667332410812378, |
| "sampling/importance_sampling_ratio/max": 1.4638419151306152, |
| "sampling/importance_sampling_ratio/mean": 0.9871161580085754, |
| "sampling/importance_sampling_ratio/min": 0.5785120725631714, |
| "sampling/sampling_logp_difference/max": 0.5472958087921143, |
| "sampling/sampling_logp_difference/mean": 0.0218554325401783, |
| "step": 31, |
| "step_time": 24.88910816120915 |
| }, |
| { |
| "clip_ratio/high_max": 0.000850055415260916, |
| "clip_ratio/high_mean": 0.000850055415260916, |
| "clip_ratio/low_mean": 9.867772071932752e-05, |
| "clip_ratio/low_min": 9.867772071932752e-05, |
| "clip_ratio/region_mean": 0.0009487331359802434, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 752.0, |
| "completions/max_terminated_length": 752.0, |
| "completions/mean_length": 173.00001525878906, |
| "completions/mean_terminated_length": 173.00001525878906, |
| "completions/min_length": 60.0, |
| "completions/min_terminated_length": 60.0, |
| "entropy": 0.20183096577723822, |
| "epoch": 0.08333333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006611433811485767, |
| "learning_rate": 1e-05, |
| "loss": 0.026979386806488037, |
| "num_tokens": 762217.0, |
| "reward": 0.2488606870174408, |
| "reward_std": 0.49546363949775696, |
| "rewards/correctness/mean": 0.3333333432674408, |
| "rewards/correctness/std": 0.47538265585899353, |
| "rewards/length_penalty/mean": -0.08447265625, |
| "rewards/length_penalty/std": 0.060732532292604446, |
| "sampling/importance_sampling_ratio/max": 1.409083604812622, |
| "sampling/importance_sampling_ratio/mean": 0.9928571581840515, |
| "sampling/importance_sampling_ratio/min": 0.5523043274879456, |
| "sampling/sampling_logp_difference/max": 0.593656063079834, |
| "sampling/sampling_logp_difference/mean": 0.013818368315696716, |
| "step": 32, |
| "step_time": 10.099375926190987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010567558007702853, |
| "clip_ratio/high_mean": 0.0010567558007702853, |
| "clip_ratio/low_mean": 0.00020370222046039999, |
| "clip_ratio/low_min": 0.00020370222046039999, |
| "clip_ratio/region_mean": 0.0012604580260813236, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1939.0, |
| "completions/max_terminated_length": 1939.0, |
| "completions/mean_length": 356.0500183105469, |
| "completions/mean_terminated_length": 356.0500183105469, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.35141467054684955, |
| "epoch": 0.0859375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010417474433779716, |
| "learning_rate": 1e-05, |
| "loss": 0.053548067808151245, |
| "num_tokens": 788430.0, |
| "reward": 0.4094808101654053, |
| "reward_std": 0.5757761001586914, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.17385253310203552, |
| "rewards/length_penalty/std": 0.16389718651771545, |
| "sampling/importance_sampling_ratio/max": 1.6124908924102783, |
| "sampling/importance_sampling_ratio/mean": 0.9880713820457458, |
| "sampling/importance_sampling_ratio/min": 0.5255787372589111, |
| "sampling/sampling_logp_difference/max": 0.6432552337646484, |
| "sampling/sampling_logp_difference/mean": 0.021633507683873177, |
| "step": 33, |
| "step_time": 23.034046452259645 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009480533820654576, |
| "clip_ratio/high_mean": 0.0009480533820654576, |
| "clip_ratio/low_mean": 0.00024225046702971062, |
| "clip_ratio/low_min": 0.00024225046702971062, |
| "clip_ratio/region_mean": 0.0011903038442445297, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1264.0, |
| "completions/mean_length": 285.16668701171875, |
| "completions/mean_terminated_length": 224.37930297851562, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.3103243013223012, |
| "epoch": 0.08854166666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005638830363750458, |
| "learning_rate": 1e-05, |
| "loss": 0.0008094119839370251, |
| "num_tokens": 810000.0, |
| "reward": 0.5940918326377869, |
| "reward_std": 0.572606086730957, |
| "rewards/correctness/mean": 0.7333333492279053, |
| "rewards/correctness/std": 0.4459485113620758, |
| "rewards/length_penalty/mean": -0.1392415314912796, |
| "rewards/length_penalty/std": 0.19301527738571167, |
| "sampling/importance_sampling_ratio/max": 1.4084635972976685, |
| "sampling/importance_sampling_ratio/mean": 0.9888424873352051, |
| "sampling/importance_sampling_ratio/min": 0.6867771148681641, |
| "sampling/sampling_logp_difference/max": 0.37574541568756104, |
| "sampling/sampling_logp_difference/mean": 0.019862888380885124, |
| "step": 34, |
| "step_time": 24.538513879058883 |
| }, |
| { |
| "clip_ratio/high_max": 0.001483886589994654, |
| "clip_ratio/high_mean": 0.001483886589994654, |
| "clip_ratio/low_mean": 0.00026628037934036303, |
| "clip_ratio/low_min": 0.00026628037934036303, |
| "clip_ratio/region_mean": 0.001750167003289486, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 857.0, |
| "completions/max_terminated_length": 857.0, |
| "completions/mean_length": 256.8333435058594, |
| "completions/mean_terminated_length": 256.8333435058594, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.28591489295164746, |
| "epoch": 0.09114583333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007422458380460739, |
| "learning_rate": 1e-05, |
| "loss": 0.00048685818910598755, |
| "num_tokens": 828970.0, |
| "reward": 0.5579264760017395, |
| "reward_std": 0.49745243787765503, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.1254069060087204, |
| "rewards/length_penalty/std": 0.07776686549186707, |
| "sampling/importance_sampling_ratio/max": 1.435230016708374, |
| "sampling/importance_sampling_ratio/mean": 0.9897229075431824, |
| "sampling/importance_sampling_ratio/min": 0.6711592674255371, |
| "sampling/sampling_logp_difference/max": 0.39874887466430664, |
| "sampling/sampling_logp_difference/mean": 0.017357412725687027, |
| "step": 35, |
| "step_time": 10.571540778968483 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009182901800765345, |
| "clip_ratio/high_mean": 0.0009182901800765345, |
| "clip_ratio/low_mean": 0.00022369111441851905, |
| "clip_ratio/low_min": 0.00022369111441851905, |
| "clip_ratio/region_mean": 0.0011419813284495224, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1777.0, |
| "completions/mean_length": 408.183349609375, |
| "completions/mean_terminated_length": 351.637939453125, |
| "completions/min_length": 118.0, |
| "completions/min_terminated_length": 118.0, |
| "entropy": 0.3397334814071655, |
| "epoch": 0.09375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007984301075339317, |
| "learning_rate": 1e-05, |
| "loss": 0.017920570448040962, |
| "num_tokens": 859101.0, |
| "reward": 0.23402507603168488, |
| "reward_std": 0.5911550521850586, |
| "rewards/correctness/mean": 0.4333333373069763, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.19930826127529144, |
| "rewards/length_penalty/std": 0.213369682431221, |
| "sampling/importance_sampling_ratio/max": 1.560028076171875, |
| "sampling/importance_sampling_ratio/mean": 0.9881030321121216, |
| "sampling/importance_sampling_ratio/min": 0.6365365982055664, |
| "sampling/sampling_logp_difference/max": 0.45171332359313965, |
| "sampling/sampling_logp_difference/mean": 0.02074795961380005, |
| "step": 36, |
| "step_time": 25.462272150907665 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013787642043704789, |
| "clip_ratio/high_mean": 0.0013787642043704789, |
| "clip_ratio/low_mean": 4.437344614416361e-05, |
| "clip_ratio/low_min": 4.437344614416361e-05, |
| "clip_ratio/region_mean": 0.0014231376505146425, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1822.0, |
| "completions/mean_length": 353.38336181640625, |
| "completions/mean_terminated_length": 264.1929931640625, |
| "completions/min_length": 122.0, |
| "completions/min_terminated_length": 122.0, |
| "entropy": 0.2575418601433436, |
| "epoch": 0.09635416666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009004107676446438, |
| "learning_rate": 1e-05, |
| "loss": 0.03392153978347778, |
| "num_tokens": 884704.0, |
| "reward": 0.5107828974723816, |
| "reward_std": 0.5746500492095947, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.17255045473575592, |
| "rewards/length_penalty/std": 0.23225854337215424, |
| "sampling/importance_sampling_ratio/max": 1.4906537532806396, |
| "sampling/importance_sampling_ratio/mean": 0.9905866980552673, |
| "sampling/importance_sampling_ratio/min": 0.6603085398674011, |
| "sampling/sampling_logp_difference/max": 0.41504812240600586, |
| "sampling/sampling_logp_difference/mean": 0.016476230695843697, |
| "step": 37, |
| "step_time": 24.73520529945381 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010135239814796175, |
| "clip_ratio/high_mean": 0.0010135239814796175, |
| "clip_ratio/low_mean": 0.0003116595641283008, |
| "clip_ratio/low_min": 0.0003116595641283008, |
| "clip_ratio/region_mean": 0.0013251835577345143, |
| "completions/clipped_ratio": 0.06666667014360428, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1675.0, |
| "completions/mean_length": 535.9500122070312, |
| "completions/mean_terminated_length": 427.9464416503906, |
| "completions/min_length": 133.0, |
| "completions/min_terminated_length": 133.0, |
| "entropy": 0.3643091519673665, |
| "epoch": 0.09895833333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012251719832420349, |
| "learning_rate": 1e-05, |
| "loss": 0.10831046104431152, |
| "num_tokens": 922851.0, |
| "reward": 0.3216390013694763, |
| "reward_std": 0.6570683121681213, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.2616943418979645, |
| "rewards/length_penalty/std": 0.25460362434387207, |
| "sampling/importance_sampling_ratio/max": 1.4309269189834595, |
| "sampling/importance_sampling_ratio/mean": 0.9871833920478821, |
| "sampling/importance_sampling_ratio/min": 0.6401387453079224, |
| "sampling/sampling_logp_difference/max": 0.44607043266296387, |
| "sampling/sampling_logp_difference/mean": 0.022042104974389076, |
| "step": 38, |
| "step_time": 25.6242256781552 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008855094395888349, |
| "clip_ratio/high_mean": 0.0008855094395888349, |
| "clip_ratio/low_mean": 0.00023092723859008402, |
| "clip_ratio/low_min": 0.00023092723859008402, |
| "clip_ratio/region_mean": 0.0011164366830295573, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1375.0, |
| "completions/mean_length": 280.45001220703125, |
| "completions/mean_terminated_length": 250.4915313720703, |
| "completions/min_length": 54.0, |
| "completions/min_terminated_length": 54.0, |
| "entropy": 0.357806995511055, |
| "epoch": 0.1015625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00958824623376131, |
| "learning_rate": 1e-05, |
| "loss": 0.025433354079723358, |
| "num_tokens": 943438.0, |
| "reward": 0.47972822189331055, |
| "reward_std": 0.5603465437889099, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.13693848252296448, |
| "rewards/length_penalty/std": 0.1538229137659073, |
| "sampling/importance_sampling_ratio/max": 1.4670376777648926, |
| "sampling/importance_sampling_ratio/mean": 0.9874992966651917, |
| "sampling/importance_sampling_ratio/min": 0.645994246006012, |
| "sampling/sampling_logp_difference/max": 0.436964750289917, |
| "sampling/sampling_logp_difference/mean": 0.021456707268953323, |
| "step": 39, |
| "step_time": 24.154640823369846 |
| }, |
| { |
| "clip_ratio/high_max": 0.001127492607338354, |
| "clip_ratio/high_mean": 0.001127492607338354, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001127492607338354, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 461.0, |
| "completions/max_terminated_length": 461.0, |
| "completions/mean_length": 149.0166778564453, |
| "completions/mean_terminated_length": 149.0166778564453, |
| "completions/min_length": 42.0, |
| "completions/min_terminated_length": 42.0, |
| "entropy": 0.21087645242611566, |
| "epoch": 0.10416666666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006187669467180967, |
| "learning_rate": 1e-05, |
| "loss": -0.004451556596904993, |
| "num_tokens": 955529.0, |
| "reward": 0.7105713486671448, |
| "reward_std": 0.4294970631599426, |
| "rewards/correctness/mean": 0.7833333611488342, |
| "rewards/correctness/std": 0.41545024514198303, |
| "rewards/length_penalty/mean": -0.07276204228401184, |
| "rewards/length_penalty/std": 0.031060174107551575, |
| "sampling/importance_sampling_ratio/max": 1.4953477382659912, |
| "sampling/importance_sampling_ratio/mean": 0.9920693039894104, |
| "sampling/importance_sampling_ratio/min": 0.7115786671638489, |
| "sampling/sampling_logp_difference/max": 0.4023587703704834, |
| "sampling/sampling_logp_difference/mean": 0.014775531366467476, |
| "step": 40, |
| "step_time": 6.007723273709416 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008233496046159416, |
| "clip_ratio/high_mean": 0.0008233496046159416, |
| "clip_ratio/low_mean": 0.00011454753500098984, |
| "clip_ratio/low_min": 0.00011454753500098984, |
| "clip_ratio/region_mean": 0.0009378971396169314, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 477.0, |
| "completions/max_terminated_length": 477.0, |
| "completions/mean_length": 151.7166748046875, |
| "completions/mean_terminated_length": 151.7166748046875, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.2801721890767415, |
| "epoch": 0.10677083333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00714639388024807, |
| "learning_rate": 1e-05, |
| "loss": 0.003442673245444894, |
| "num_tokens": 968502.0, |
| "reward": 0.5259196162223816, |
| "reward_std": 0.5126462578773499, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.07408040016889572, |
| "rewards/length_penalty/std": 0.03312588483095169, |
| "sampling/importance_sampling_ratio/max": 1.4574203491210938, |
| "sampling/importance_sampling_ratio/mean": 0.9899853467941284, |
| "sampling/importance_sampling_ratio/min": 0.6682142615318298, |
| "sampling/sampling_logp_difference/max": 0.4031463861465454, |
| "sampling/sampling_logp_difference/mean": 0.01854798197746277, |
| "step": 41, |
| "step_time": 6.347758481046185 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008199614191350216, |
| "clip_ratio/high_mean": 0.0008199614191350216, |
| "clip_ratio/low_mean": 9.921489496870588e-05, |
| "clip_ratio/low_min": 9.921489496870588e-05, |
| "clip_ratio/region_mean": 0.0009191763286556428, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1759.0, |
| "completions/mean_length": 434.8166809082031, |
| "completions/mean_terminated_length": 407.4745788574219, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.4098270038763682, |
| "epoch": 0.109375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011212454177439213, |
| "learning_rate": 1e-05, |
| "loss": 0.03022373840212822, |
| "num_tokens": 999461.0, |
| "reward": 0.5376871824264526, |
| "reward_std": 0.6064243316650391, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.21231283247470856, |
| "rewards/length_penalty/std": 0.22932446002960205, |
| "sampling/importance_sampling_ratio/max": 1.7145400047302246, |
| "sampling/importance_sampling_ratio/mean": 0.9863914251327515, |
| "sampling/importance_sampling_ratio/min": 0.6030144691467285, |
| "sampling/sampling_logp_difference/max": 0.5391448736190796, |
| "sampling/sampling_logp_difference/mean": 0.023289671167731285, |
| "step": 42, |
| "step_time": 26.201360404258594 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010472298308741301, |
| "clip_ratio/high_mean": 0.0010472298308741301, |
| "clip_ratio/low_mean": 0.0002770637341503364, |
| "clip_ratio/low_min": 0.0002770637341503364, |
| "clip_ratio/region_mean": 0.001324293582001701, |
| "completions/clipped_ratio": 0.0833333358168602, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1712.0, |
| "completions/mean_length": 507.10003662109375, |
| "completions/mean_terminated_length": 367.0181579589844, |
| "completions/min_length": 67.0, |
| "completions/min_terminated_length": 67.0, |
| "entropy": 0.31672103703022003, |
| "epoch": 0.11197916666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016588151454925537, |
| "learning_rate": 1e-05, |
| "loss": 0.0822218656539917, |
| "num_tokens": 1033587.0, |
| "reward": 0.4523926079273224, |
| "reward_std": 0.6258042454719543, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.24760742485523224, |
| "rewards/length_penalty/std": 0.2824883759021759, |
| "sampling/importance_sampling_ratio/max": 1.8998955488204956, |
| "sampling/importance_sampling_ratio/mean": 0.9888049960136414, |
| "sampling/importance_sampling_ratio/min": 0.5521621108055115, |
| "sampling/sampling_logp_difference/max": 0.6417989730834961, |
| "sampling/sampling_logp_difference/mean": 0.019485708326101303, |
| "step": 43, |
| "step_time": 24.33553890278563 |
| }, |
| { |
| "clip_ratio/high_max": 0.0017707225245734055, |
| "clip_ratio/high_mean": 0.0017707225245734055, |
| "clip_ratio/low_mean": 9.580219436126451e-05, |
| "clip_ratio/low_min": 9.580219436126451e-05, |
| "clip_ratio/region_mean": 0.0018665247092333932, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1729.0, |
| "completions/mean_length": 404.7166748046875, |
| "completions/mean_terminated_length": 348.0517272949219, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.3011014685034752, |
| "epoch": 0.11458333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010350089520215988, |
| "learning_rate": 1e-05, |
| "loss": 0.026937788352370262, |
| "num_tokens": 1062790.0, |
| "reward": 0.5023844838142395, |
| "reward_std": 0.5291100740432739, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.19761556386947632, |
| "rewards/length_penalty/std": 0.22229313850402832, |
| "sampling/importance_sampling_ratio/max": 1.4671045541763306, |
| "sampling/importance_sampling_ratio/mean": 0.9886985421180725, |
| "sampling/importance_sampling_ratio/min": 0.6522778272628784, |
| "sampling/sampling_logp_difference/max": 0.42728471755981445, |
| "sampling/sampling_logp_difference/mean": 0.018706969916820526, |
| "step": 44, |
| "step_time": 23.952977614942938 |
| }, |
| { |
| "clip_ratio/high_max": 0.00042857573134824634, |
| "clip_ratio/high_mean": 0.00042857573134824634, |
| "clip_ratio/low_mean": 0.00022722109376142421, |
| "clip_ratio/low_min": 0.00022722109376142421, |
| "clip_ratio/region_mean": 0.0006557968251096705, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 776.0, |
| "completions/max_terminated_length": 776.0, |
| "completions/mean_length": 141.5, |
| "completions/mean_terminated_length": 141.5, |
| "completions/min_length": 66.0, |
| "completions/min_terminated_length": 66.0, |
| "entropy": 0.2517914300163587, |
| "epoch": 0.1171875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005560688208788633, |
| "learning_rate": 1e-05, |
| "loss": 0.00028455094434320927, |
| "num_tokens": 1076640.0, |
| "reward": 0.7142415642738342, |
| "reward_std": 0.44235333800315857, |
| "rewards/correctness/mean": 0.7833333611488342, |
| "rewards/correctness/std": 0.41545021533966064, |
| "rewards/length_penalty/mean": -0.069091796875, |
| "rewards/length_penalty/std": 0.05174348130822182, |
| "sampling/importance_sampling_ratio/max": 1.6184712648391724, |
| "sampling/importance_sampling_ratio/mean": 0.9908018112182617, |
| "sampling/importance_sampling_ratio/min": 0.6661750674247742, |
| "sampling/sampling_logp_difference/max": 0.48148202896118164, |
| "sampling/sampling_logp_difference/mean": 0.017505010589957237, |
| "step": 45, |
| "step_time": 10.239898391533643 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011402330710552633, |
| "clip_ratio/high_mean": 0.0011402330710552633, |
| "clip_ratio/low_mean": 0.00020916750994122898, |
| "clip_ratio/low_min": 0.00020916750994122898, |
| "clip_ratio/region_mean": 0.001349400554317981, |
| "completions/clipped_ratio": 0.13333334028720856, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1575.0, |
| "completions/mean_length": 647.36669921875, |
| "completions/mean_terminated_length": 431.8846435546875, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.4430996775627136, |
| "epoch": 0.11979166666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009722047485411167, |
| "learning_rate": 1e-05, |
| "loss": -0.0408431775867939, |
| "num_tokens": 1121172.0, |
| "reward": 0.3172363340854645, |
| "reward_std": 0.7371562719345093, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.3160969913005829, |
| "rewards/length_penalty/std": 0.32975026965141296, |
| "sampling/importance_sampling_ratio/max": 2.2945001125335693, |
| "sampling/importance_sampling_ratio/mean": 0.9847511053085327, |
| "sampling/importance_sampling_ratio/min": 0.6078668236732483, |
| "sampling/sampling_logp_difference/max": 0.8305149078369141, |
| "sampling/sampling_logp_difference/mean": 0.025009581819176674, |
| "step": 46, |
| "step_time": 26.005909194936976 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012073956701594095, |
| "clip_ratio/high_mean": 0.0012073956701594095, |
| "clip_ratio/low_mean": 3.7605294589108475e-05, |
| "clip_ratio/low_min": 3.7605294589108475e-05, |
| "clip_ratio/region_mean": 0.0012450009623231988, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1538.0, |
| "completions/max_terminated_length": 1538.0, |
| "completions/mean_length": 316.3000183105469, |
| "completions/mean_terminated_length": 316.3000183105469, |
| "completions/min_length": 61.0, |
| "completions/min_terminated_length": 61.0, |
| "entropy": 0.30036087334156036, |
| "epoch": 0.12239583333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008779074996709824, |
| "learning_rate": 1e-05, |
| "loss": 0.005721051711589098, |
| "num_tokens": 1143870.0, |
| "reward": 0.4288899898529053, |
| "reward_std": 0.5387234091758728, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.15444335341453552, |
| "rewards/length_penalty/std": 0.13659639656543732, |
| "sampling/importance_sampling_ratio/max": 1.428618311882019, |
| "sampling/importance_sampling_ratio/mean": 0.9887758493423462, |
| "sampling/importance_sampling_ratio/min": 0.648135781288147, |
| "sampling/sampling_logp_difference/max": 0.4336550235748291, |
| "sampling/sampling_logp_difference/mean": 0.01943063922226429, |
| "step": 47, |
| "step_time": 17.67896143696271 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003499378120371451, |
| "clip_ratio/high_mean": 0.0003499378120371451, |
| "clip_ratio/low_mean": 0.00016028637279911587, |
| "clip_ratio/low_min": 0.00016028637279911587, |
| "clip_ratio/region_mean": 0.000510224184836261, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1451.0, |
| "completions/max_terminated_length": 1451.0, |
| "completions/mean_length": 308.5833435058594, |
| "completions/mean_terminated_length": 308.5833435058594, |
| "completions/min_length": 44.0, |
| "completions/min_terminated_length": 44.0, |
| "entropy": 0.27901175369819003, |
| "epoch": 0.125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008123793639242649, |
| "learning_rate": 1e-05, |
| "loss": 0.01864437572658062, |
| "num_tokens": 1167955.0, |
| "reward": 0.48265790939331055, |
| "reward_std": 0.5894247889518738, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.1506754606962204, |
| "rewards/length_penalty/std": 0.14377768337726593, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9899443984031677, |
| "sampling/importance_sampling_ratio/min": 0.6312875747680664, |
| "sampling/sampling_logp_difference/max": 1.282071828842163, |
| "sampling/sampling_logp_difference/mean": 0.018121350556612015, |
| "step": 48, |
| "step_time": 17.112844260176644 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009603448124835268, |
| "clip_ratio/high_mean": 0.0009603448124835268, |
| "clip_ratio/low_mean": 7.887679385021329e-05, |
| "clip_ratio/low_min": 7.887679385021329e-05, |
| "clip_ratio/region_mean": 0.00103922160633374, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 897.0, |
| "completions/mean_length": 259.1500244140625, |
| "completions/mean_terminated_length": 228.83050537109375, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.26935919870932895, |
| "epoch": 0.12760416666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007601945661008358, |
| "learning_rate": 1e-05, |
| "loss": 0.017746908590197563, |
| "num_tokens": 1188384.0, |
| "reward": 0.4401285946369171, |
| "reward_std": 0.5509795546531677, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.12653808295726776, |
| "rewards/length_penalty/std": 0.1379641592502594, |
| "sampling/importance_sampling_ratio/max": 1.4395726919174194, |
| "sampling/importance_sampling_ratio/mean": 0.9891641139984131, |
| "sampling/importance_sampling_ratio/min": 0.6558746695518494, |
| "sampling/sampling_logp_difference/max": 0.4217855930328369, |
| "sampling/sampling_logp_difference/mean": 0.018416719511151314, |
| "step": 49, |
| "step_time": 23.047351316083223 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004951291290732721, |
| "clip_ratio/high_mean": 0.0004951291290732721, |
| "clip_ratio/low_mean": 0.00016483282282327613, |
| "clip_ratio/low_min": 0.00016483282282327613, |
| "clip_ratio/region_mean": 0.0006599619470459098, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1239.0, |
| "completions/max_terminated_length": 1239.0, |
| "completions/mean_length": 285.7166748046875, |
| "completions/mean_terminated_length": 285.7166748046875, |
| "completions/min_length": 91.0, |
| "completions/min_terminated_length": 91.0, |
| "entropy": 0.24318194389343262, |
| "epoch": 0.13020833333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007684387266635895, |
| "learning_rate": 1e-05, |
| "loss": 0.01248091645538807, |
| "num_tokens": 1209877.0, |
| "reward": 0.42715659737586975, |
| "reward_std": 0.5595811605453491, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971747398376465, |
| "rewards/length_penalty/mean": -0.13951009511947632, |
| "rewards/length_penalty/std": 0.11080927401781082, |
| "sampling/importance_sampling_ratio/max": 1.3555326461791992, |
| "sampling/importance_sampling_ratio/mean": 0.9910274744033813, |
| "sampling/importance_sampling_ratio/min": 0.6758385896682739, |
| "sampling/sampling_logp_difference/max": 0.39180099964141846, |
| "sampling/sampling_logp_difference/mean": 0.016018925234675407, |
| "step": 50, |
| "step_time": 14.20961848529987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008978430705610663, |
| "clip_ratio/high_mean": 0.0008978430705610663, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008978430705610663, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 831.0, |
| "completions/max_terminated_length": 831.0, |
| "completions/mean_length": 190.8000030517578, |
| "completions/mean_terminated_length": 190.8000030517578, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.23509576171636581, |
| "epoch": 0.1328125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006647464353591204, |
| "learning_rate": 1e-05, |
| "loss": -0.008888879790902138, |
| "num_tokens": 1225115.0, |
| "reward": 0.5735026597976685, |
| "reward_std": 0.48649072647094727, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.09316406399011612, |
| "rewards/length_penalty/std": 0.05234139785170555, |
| "sampling/importance_sampling_ratio/max": 1.2898887395858765, |
| "sampling/importance_sampling_ratio/mean": 0.9915088415145874, |
| "sampling/importance_sampling_ratio/min": 0.6340304017066956, |
| "sampling/sampling_logp_difference/max": 0.4556584358215332, |
| "sampling/sampling_logp_difference/mean": 0.016132386401295662, |
| "step": 51, |
| "step_time": 10.349144363077357 |
| }, |
| { |
| "clip_ratio/high_max": 0.001242596423253417, |
| "clip_ratio/high_mean": 0.001242596423253417, |
| "clip_ratio/low_mean": 7.255841046571732e-05, |
| "clip_ratio/low_min": 7.255841046571732e-05, |
| "clip_ratio/region_mean": 0.0013151548337191343, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 623.0, |
| "completions/max_terminated_length": 623.0, |
| "completions/mean_length": 190.0166778564453, |
| "completions/mean_terminated_length": 190.0166778564453, |
| "completions/min_length": 87.0, |
| "completions/min_terminated_length": 87.0, |
| "entropy": 0.29815925161043805, |
| "epoch": 0.13541666666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006168007850646973, |
| "learning_rate": 1e-05, |
| "loss": -0.008402797393500805, |
| "num_tokens": 1240306.0, |
| "reward": 0.6572184562683105, |
| "reward_std": 0.4351221024990082, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.09278157353401184, |
| "rewards/length_penalty/std": 0.03590512275695801, |
| "sampling/importance_sampling_ratio/max": 1.3934484720230103, |
| "sampling/importance_sampling_ratio/mean": 0.9897996783256531, |
| "sampling/importance_sampling_ratio/min": 0.6509549617767334, |
| "sampling/sampling_logp_difference/max": 0.42931485176086426, |
| "sampling/sampling_logp_difference/mean": 0.01972256973385811, |
| "step": 52, |
| "step_time": 7.439815539866686 |
| }, |
| { |
| "clip_ratio/high_max": 0.001043767377268523, |
| "clip_ratio/high_mean": 0.001043767377268523, |
| "clip_ratio/low_mean": 0.0001336474670097232, |
| "clip_ratio/low_min": 0.0001336474670097232, |
| "clip_ratio/region_mean": 0.0011774148442782462, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1946.0, |
| "completions/mean_length": 337.8000183105469, |
| "completions/mean_terminated_length": 308.8135681152344, |
| "completions/min_length": 42.0, |
| "completions/min_terminated_length": 42.0, |
| "entropy": 0.40103500584761304, |
| "epoch": 0.13802083333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012700427323579788, |
| "learning_rate": 1e-05, |
| "loss": 0.0148848881945014, |
| "num_tokens": 1265104.0, |
| "reward": 0.6517252922058105, |
| "reward_std": 0.5710904598236084, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.16494140028953552, |
| "rewards/length_penalty/std": 0.23197756707668304, |
| "sampling/importance_sampling_ratio/max": 1.5886998176574707, |
| "sampling/importance_sampling_ratio/mean": 0.9859074354171753, |
| "sampling/importance_sampling_ratio/min": 0.6386861801147461, |
| "sampling/sampling_logp_difference/max": 0.4629160165786743, |
| "sampling/sampling_logp_difference/mean": 0.023820485919713974, |
| "step": 53, |
| "step_time": 23.753358610672876 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011367613672822092, |
| "clip_ratio/high_mean": 0.0011367613672822092, |
| "clip_ratio/low_mean": 0.00010521885512086253, |
| "clip_ratio/low_min": 0.00010521885512086253, |
| "clip_ratio/region_mean": 0.0012419802321043487, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1376.0, |
| "completions/max_terminated_length": 1376.0, |
| "completions/mean_length": 180.20001220703125, |
| "completions/mean_terminated_length": 180.20001220703125, |
| "completions/min_length": 26.0, |
| "completions/min_terminated_length": 26.0, |
| "entropy": 0.25668895492951077, |
| "epoch": 0.140625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006395912729203701, |
| "learning_rate": 1e-05, |
| "loss": 0.0007007146487012506, |
| "num_tokens": 1279596.0, |
| "reward": 0.5286784172058105, |
| "reward_std": 0.5428489446640015, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.08798827975988388, |
| "rewards/length_penalty/std": 0.1015106737613678, |
| "sampling/importance_sampling_ratio/max": 1.5878007411956787, |
| "sampling/importance_sampling_ratio/mean": 0.9908829927444458, |
| "sampling/importance_sampling_ratio/min": 0.35455605387687683, |
| "sampling/sampling_logp_difference/max": 1.036888837814331, |
| "sampling/sampling_logp_difference/mean": 0.017583977431058884, |
| "step": 54, |
| "step_time": 15.143645148025826 |
| }, |
| { |
| "clip_ratio/high_max": 0.00025663612177595496, |
| "clip_ratio/high_mean": 0.00025663612177595496, |
| "clip_ratio/low_mean": 0.00015472400991711766, |
| "clip_ratio/low_min": 0.00015472400991711766, |
| "clip_ratio/region_mean": 0.0004113601268424342, |
| "completions/clipped_ratio": 0.11666667461395264, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1819.0, |
| "completions/mean_length": 376.2166748046875, |
| "completions/mean_terminated_length": 155.41510009765625, |
| "completions/min_length": 51.0, |
| "completions/min_terminated_length": 51.0, |
| "entropy": 0.3149576584498088, |
| "epoch": 0.14322916666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01507708989083767, |
| "learning_rate": 1e-05, |
| "loss": -0.0034385398030281067, |
| "num_tokens": 1306409.0, |
| "reward": 0.6329671740531921, |
| "reward_std": 0.667377769947052, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.18369954824447632, |
| "rewards/length_penalty/std": 0.3198634088039398, |
| "sampling/importance_sampling_ratio/max": 1.4602930545806885, |
| "sampling/importance_sampling_ratio/mean": 0.9883601069450378, |
| "sampling/importance_sampling_ratio/min": 0.6380261182785034, |
| "sampling/sampling_logp_difference/max": 0.44937610626220703, |
| "sampling/sampling_logp_difference/mean": 0.020671239122748375, |
| "step": 55, |
| "step_time": 23.81720179039985 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014060426037758589, |
| "clip_ratio/high_mean": 0.0014060426037758589, |
| "clip_ratio/low_mean": 8.169935123684506e-05, |
| "clip_ratio/low_min": 8.169935123684506e-05, |
| "clip_ratio/region_mean": 0.001487741955012704, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1646.0, |
| "completions/max_terminated_length": 1646.0, |
| "completions/mean_length": 228.15000915527344, |
| "completions/mean_terminated_length": 228.15000915527344, |
| "completions/min_length": 72.0, |
| "completions/min_terminated_length": 72.0, |
| "entropy": 0.32602732876936596, |
| "epoch": 0.14583333333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008335023187100887, |
| "learning_rate": 1e-05, |
| "loss": -0.029388360679149628, |
| "num_tokens": 1323728.0, |
| "reward": 0.40526533126831055, |
| "reward_std": 0.4930865168571472, |
| "rewards/correctness/mean": 0.5166666507720947, |
| "rewards/correctness/std": 0.5039393305778503, |
| "rewards/length_penalty/mean": -0.11140136420726776, |
| "rewards/length_penalty/std": 0.1095164492726326, |
| "sampling/importance_sampling_ratio/max": 1.4040553569793701, |
| "sampling/importance_sampling_ratio/mean": 0.9880329966545105, |
| "sampling/importance_sampling_ratio/min": 0.6747580170631409, |
| "sampling/sampling_logp_difference/max": 0.3934011459350586, |
| "sampling/sampling_logp_difference/mean": 0.02050255611538887, |
| "step": 56, |
| "step_time": 17.96976805315353 |
| }, |
| { |
| "clip_ratio/high_max": 0.000578798086886915, |
| "clip_ratio/high_mean": 0.000578798086886915, |
| "clip_ratio/low_mean": 0.0001190476177725941, |
| "clip_ratio/low_min": 0.0001190476177725941, |
| "clip_ratio/region_mean": 0.0006978457046595091, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 784.0, |
| "completions/max_terminated_length": 784.0, |
| "completions/mean_length": 231.1333465576172, |
| "completions/mean_terminated_length": 231.1333465576172, |
| "completions/min_length": 77.0, |
| "completions/min_terminated_length": 77.0, |
| "entropy": 0.27178917328516644, |
| "epoch": 0.1484375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0058025578036904335, |
| "learning_rate": 1e-05, |
| "loss": -0.008416585624217987, |
| "num_tokens": 1342806.0, |
| "reward": 0.47047528624534607, |
| "reward_std": 0.5530902743339539, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.11285807192325592, |
| "rewards/length_penalty/std": 0.09149719774723053, |
| "sampling/importance_sampling_ratio/max": 1.393749475479126, |
| "sampling/importance_sampling_ratio/mean": 0.9902473092079163, |
| "sampling/importance_sampling_ratio/min": 0.6727088689804077, |
| "sampling/sampling_logp_difference/max": 0.3964426517486572, |
| "sampling/sampling_logp_difference/mean": 0.017496805638074875, |
| "step": 57, |
| "step_time": 10.656542683253065 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010802776378113776, |
| "clip_ratio/high_mean": 0.0010802776378113776, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010802776378113776, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 661.0, |
| "completions/max_terminated_length": 661.0, |
| "completions/mean_length": 182.6333465576172, |
| "completions/mean_terminated_length": 182.6333465576172, |
| "completions/min_length": 66.0, |
| "completions/min_terminated_length": 66.0, |
| "entropy": 0.24240783601999283, |
| "epoch": 0.15104166666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007706186734139919, |
| "learning_rate": 1e-05, |
| "loss": -0.011561744846403599, |
| "num_tokens": 1357304.0, |
| "reward": 0.6941569447517395, |
| "reward_std": 0.4358154237270355, |
| "rewards/correctness/mean": 0.7833333611488342, |
| "rewards/correctness/std": 0.41545024514198303, |
| "rewards/length_penalty/mean": -0.08917643129825592, |
| "rewards/length_penalty/std": 0.059183269739151, |
| "sampling/importance_sampling_ratio/max": 1.334013819694519, |
| "sampling/importance_sampling_ratio/mean": 0.9917068481445312, |
| "sampling/importance_sampling_ratio/min": 0.690686821937561, |
| "sampling/sampling_logp_difference/max": 0.3700687885284424, |
| "sampling/sampling_logp_difference/mean": 0.0160524882376194, |
| "step": 58, |
| "step_time": 7.915996998315677 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011899647312626864, |
| "clip_ratio/high_mean": 0.0011899647312626864, |
| "clip_ratio/low_mean": 0.0001323101023444906, |
| "clip_ratio/low_min": 0.0001323101023444906, |
| "clip_ratio/region_mean": 0.0013222748384578153, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1948.0, |
| "completions/mean_length": 470.70001220703125, |
| "completions/mean_terminated_length": 416.3103332519531, |
| "completions/min_length": 101.0, |
| "completions/min_terminated_length": 101.0, |
| "entropy": 0.34853876133759815, |
| "epoch": 0.15364583333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012271206825971603, |
| "learning_rate": 1e-05, |
| "loss": 0.04311361908912659, |
| "num_tokens": 1391626.0, |
| "reward": 0.38683271408081055, |
| "reward_std": 0.629518985748291, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.22983399033546448, |
| "rewards/length_penalty/std": 0.2866095304489136, |
| "sampling/importance_sampling_ratio/max": 1.4130703210830688, |
| "sampling/importance_sampling_ratio/mean": 0.9873179197311401, |
| "sampling/importance_sampling_ratio/min": 0.6609278321266174, |
| "sampling/sampling_logp_difference/max": 0.4141106605529785, |
| "sampling/sampling_logp_difference/mean": 0.021251151338219643, |
| "step": 59, |
| "step_time": 24.9323958971072 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009601690204969296, |
| "clip_ratio/high_mean": 0.0009601690204969296, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0009601690204969296, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 857.0, |
| "completions/max_terminated_length": 857.0, |
| "completions/mean_length": 218.98333740234375, |
| "completions/mean_terminated_length": 218.98333740234375, |
| "completions/min_length": 59.0, |
| "completions/min_terminated_length": 59.0, |
| "entropy": 0.3429804593324661, |
| "epoch": 0.15625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008442915044724941, |
| "learning_rate": 1e-05, |
| "loss": 0.01635800488293171, |
| "num_tokens": 1408775.0, |
| "reward": 0.5597412586212158, |
| "reward_std": 0.5319768786430359, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.10692545771598816, |
| "rewards/length_penalty/std": 0.08298203349113464, |
| "sampling/importance_sampling_ratio/max": 1.4650242328643799, |
| "sampling/importance_sampling_ratio/mean": 0.9881577491760254, |
| "sampling/importance_sampling_ratio/min": 0.6473858952522278, |
| "sampling/sampling_logp_difference/max": 0.4348127841949463, |
| "sampling/sampling_logp_difference/mean": 0.02056148275732994, |
| "step": 60, |
| "step_time": 10.053480867063627 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007526695456666251, |
| "clip_ratio/high_mean": 0.0007526695456666251, |
| "clip_ratio/low_mean": 0.00010815487864116828, |
| "clip_ratio/low_min": 0.00010815487864116828, |
| "clip_ratio/region_mean": 0.0008608244243077934, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 497.0, |
| "completions/max_terminated_length": 497.0, |
| "completions/mean_length": 157.18333435058594, |
| "completions/mean_terminated_length": 157.18333435058594, |
| "completions/min_length": 60.0, |
| "completions/min_terminated_length": 60.0, |
| "entropy": 0.2377120405435562, |
| "epoch": 0.15885416666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005714723374694586, |
| "learning_rate": 1e-05, |
| "loss": -0.0016698737163096666, |
| "num_tokens": 1423156.0, |
| "reward": 0.7899170517921448, |
| "reward_std": 0.34263694286346436, |
| "rewards/correctness/mean": 0.8666666746139526, |
| "rewards/correctness/std": 0.34280335903167725, |
| "rewards/length_penalty/mean": -0.07674967497587204, |
| "rewards/length_penalty/std": 0.041604459285736084, |
| "sampling/importance_sampling_ratio/max": 1.4403636455535889, |
| "sampling/importance_sampling_ratio/mean": 0.9908574223518372, |
| "sampling/importance_sampling_ratio/min": 0.6551518440246582, |
| "sampling/sampling_logp_difference/max": 0.42288827896118164, |
| "sampling/sampling_logp_difference/mean": 0.016409065574407578, |
| "step": 61, |
| "step_time": 6.808671161998063 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010539952927501872, |
| "clip_ratio/high_mean": 0.0010539952927501872, |
| "clip_ratio/low_mean": 0.00024715747955876094, |
| "clip_ratio/low_min": 0.00024715747955876094, |
| "clip_ratio/region_mean": 0.0013011527286532025, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1965.0, |
| "completions/mean_length": 616.6000366210938, |
| "completions/mean_terminated_length": 457.5555725097656, |
| "completions/min_length": 63.0, |
| "completions/min_terminated_length": 63.0, |
| "entropy": 0.39568065603574115, |
| "epoch": 0.16145833333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0071590072475373745, |
| "learning_rate": 1e-05, |
| "loss": 0.01634836196899414, |
| "num_tokens": 1464452.0, |
| "reward": 0.19892579317092896, |
| "reward_std": 0.7698237895965576, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5042194724082947, |
| "rewards/length_penalty/mean": -0.30107420682907104, |
| "rewards/length_penalty/std": 0.3292204737663269, |
| "sampling/importance_sampling_ratio/max": 2.1035425662994385, |
| "sampling/importance_sampling_ratio/mean": 0.9861966967582703, |
| "sampling/importance_sampling_ratio/min": 0.6485047936439514, |
| "sampling/sampling_logp_difference/max": 0.743622899055481, |
| "sampling/sampling_logp_difference/mean": 0.02334846556186676, |
| "step": 62, |
| "step_time": 25.894538902910426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008741852531481223, |
| "clip_ratio/high_mean": 0.0008741852531481223, |
| "clip_ratio/low_mean": 0.00017396594679060703, |
| "clip_ratio/low_min": 0.00017396594679060703, |
| "clip_ratio/region_mean": 0.001048151195088091, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1617.0, |
| "completions/mean_length": 368.9666748046875, |
| "completions/mean_terminated_length": 340.50848388671875, |
| "completions/min_length": 101.0, |
| "completions/min_terminated_length": 101.0, |
| "entropy": 0.3188506488998731, |
| "epoch": 0.1640625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010556783527135849, |
| "learning_rate": 1e-05, |
| "loss": 0.011296657845377922, |
| "num_tokens": 1493920.0, |
| "reward": 0.4365071952342987, |
| "reward_std": 0.5722631216049194, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014302253723, |
| "rewards/length_penalty/mean": -0.18015950918197632, |
| "rewards/length_penalty/std": 0.18029214441776276, |
| "sampling/importance_sampling_ratio/max": 1.459763526916504, |
| "sampling/importance_sampling_ratio/mean": 0.9886131882667542, |
| "sampling/importance_sampling_ratio/min": 0.5670296549797058, |
| "sampling/sampling_logp_difference/max": 0.5673435926437378, |
| "sampling/sampling_logp_difference/mean": 0.020048219710588455, |
| "step": 63, |
| "step_time": 24.552927787415683 |
| }, |
| { |
| "clip_ratio/high_max": 0.000782708356079335, |
| "clip_ratio/high_mean": 0.000782708356079335, |
| "clip_ratio/low_mean": 0.00019008784147445112, |
| "clip_ratio/low_min": 0.00019008784147445112, |
| "clip_ratio/region_mean": 0.0009727962218069782, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 814.0, |
| "completions/mean_length": 236.30001831054688, |
| "completions/mean_terminated_length": 173.8275909423828, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.2920355275273323, |
| "epoch": 0.16666666666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007897410541772842, |
| "learning_rate": 1e-05, |
| "loss": 0.032285742461681366, |
| "num_tokens": 1513128.0, |
| "reward": 0.6179525256156921, |
| "reward_std": 0.5583769679069519, |
| "rewards/correctness/mean": 0.7333333492279053, |
| "rewards/correctness/std": 0.4459484815597534, |
| "rewards/length_penalty/mean": -0.11538086086511612, |
| "rewards/length_penalty/std": 0.17788098752498627, |
| "sampling/importance_sampling_ratio/max": 1.4502114057540894, |
| "sampling/importance_sampling_ratio/mean": 0.9896523356437683, |
| "sampling/importance_sampling_ratio/min": 0.6468266248703003, |
| "sampling/sampling_logp_difference/max": 0.43567705154418945, |
| "sampling/sampling_logp_difference/mean": 0.01946558617055416, |
| "step": 64, |
| "step_time": 23.370976716047153 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004815076439020534, |
| "clip_ratio/high_mean": 0.0004815076439020534, |
| "clip_ratio/low_mean": 0.00013826467087104297, |
| "clip_ratio/low_min": 0.00013826467087104297, |
| "clip_ratio/region_mean": 0.0006197723002211811, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1892.0, |
| "completions/mean_length": 490.66668701171875, |
| "completions/mean_terminated_length": 436.96551513671875, |
| "completions/min_length": 108.0, |
| "completions/min_terminated_length": 108.0, |
| "entropy": 0.29405147830645245, |
| "epoch": 0.16927083333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009703584015369415, |
| "learning_rate": 1e-05, |
| "loss": 0.010801786556839943, |
| "num_tokens": 1549038.0, |
| "reward": 0.21041667461395264, |
| "reward_std": 0.6193122863769531, |
| "rewards/correctness/mean": 0.44999998807907104, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.2395833283662796, |
| "rewards/length_penalty/std": 0.2484065741300583, |
| "sampling/importance_sampling_ratio/max": 1.4595575332641602, |
| "sampling/importance_sampling_ratio/mean": 0.9889456629753113, |
| "sampling/importance_sampling_ratio/min": 0.547646164894104, |
| "sampling/sampling_logp_difference/max": 0.602125883102417, |
| "sampling/sampling_logp_difference/mean": 0.01884142868220806, |
| "step": 65, |
| "step_time": 25.55540833226405 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013296695542521775, |
| "clip_ratio/high_mean": 0.0013296695542521775, |
| "clip_ratio/low_mean": 0.0001034665716967235, |
| "clip_ratio/low_min": 0.0001034665716967235, |
| "clip_ratio/region_mean": 0.0014331361162476242, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1573.0, |
| "completions/max_terminated_length": 1573.0, |
| "completions/mean_length": 274.5, |
| "completions/mean_terminated_length": 274.5, |
| "completions/min_length": 60.0, |
| "completions/min_terminated_length": 60.0, |
| "entropy": 0.4110206837455432, |
| "epoch": 0.171875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007834387011826038, |
| "learning_rate": 1e-05, |
| "loss": -0.0021034418605268, |
| "num_tokens": 1569548.0, |
| "reward": 0.18263347446918488, |
| "reward_std": 0.5500053763389587, |
| "rewards/correctness/mean": 0.3166666626930237, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.134033203125, |
| "rewards/length_penalty/std": 0.17918384075164795, |
| "sampling/importance_sampling_ratio/max": 1.5589468479156494, |
| "sampling/importance_sampling_ratio/mean": 0.9857889413833618, |
| "sampling/importance_sampling_ratio/min": 0.6786117553710938, |
| "sampling/sampling_logp_difference/max": 0.44401049613952637, |
| "sampling/sampling_logp_difference/mean": 0.02443668246269226, |
| "step": 66, |
| "step_time": 18.516885918099433 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004021540662506595, |
| "clip_ratio/high_mean": 0.0004021540662506595, |
| "clip_ratio/low_mean": 0.00028926336866182584, |
| "clip_ratio/low_min": 0.00028926336866182584, |
| "clip_ratio/region_mean": 0.0006914174494644006, |
| "completions/clipped_ratio": 0.15000000596046448, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1838.0, |
| "completions/mean_length": 652.9000244140625, |
| "completions/mean_terminated_length": 406.7059020996094, |
| "completions/min_length": 60.0, |
| "completions/min_terminated_length": 60.0, |
| "entropy": 0.3308039257923762, |
| "epoch": 0.17447916666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013821459375321865, |
| "learning_rate": 1e-05, |
| "loss": -0.006678280420601368, |
| "num_tokens": 1613372.0, |
| "reward": 0.16453450918197632, |
| "reward_std": 0.7593758702278137, |
| "rewards/correctness/mean": 0.4833333194255829, |
| "rewards/correctness/std": 0.5039392709732056, |
| "rewards/length_penalty/mean": -0.31879884004592896, |
| "rewards/length_penalty/std": 0.34853091835975647, |
| "sampling/importance_sampling_ratio/max": 1.9533724784851074, |
| "sampling/importance_sampling_ratio/mean": 0.9883660078048706, |
| "sampling/importance_sampling_ratio/min": 0.49354442954063416, |
| "sampling/sampling_logp_difference/max": 0.7061424255371094, |
| "sampling/sampling_logp_difference/mean": 0.02013975940644741, |
| "step": 67, |
| "step_time": 25.266131084645167 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014806146112581093, |
| "clip_ratio/high_mean": 0.0014806146112581093, |
| "clip_ratio/low_mean": 0.00010515246928359072, |
| "clip_ratio/low_min": 0.00010515246928359072, |
| "clip_ratio/region_mean": 0.0015857670999442537, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 382.0, |
| "completions/max_terminated_length": 382.0, |
| "completions/mean_length": 164.25001525878906, |
| "completions/mean_terminated_length": 164.25001525878906, |
| "completions/min_length": 59.0, |
| "completions/min_terminated_length": 59.0, |
| "entropy": 0.23676933348178864, |
| "epoch": 0.17708333333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008019492961466312, |
| "learning_rate": 1e-05, |
| "loss": 0.014318232424557209, |
| "num_tokens": 1626787.0, |
| "reward": 0.6197998523712158, |
| "reward_std": 0.48550280928611755, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.0802001953125, |
| "rewards/length_penalty/std": 0.03402286767959595, |
| "sampling/importance_sampling_ratio/max": 1.3420710563659668, |
| "sampling/importance_sampling_ratio/mean": 0.9908727407455444, |
| "sampling/importance_sampling_ratio/min": 0.676866352558136, |
| "sampling/sampling_logp_difference/max": 0.39028143882751465, |
| "sampling/sampling_logp_difference/mean": 0.01646577939391136, |
| "step": 68, |
| "step_time": 5.328206725884229 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007667694881092757, |
| "clip_ratio/high_mean": 0.0007667694881092757, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007667694881092757, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 755.0, |
| "completions/max_terminated_length": 755.0, |
| "completions/mean_length": 181.98333740234375, |
| "completions/mean_terminated_length": 181.98333740234375, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.2996346205472946, |
| "epoch": 0.1796875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008045333437621593, |
| "learning_rate": 1e-05, |
| "loss": 0.022082606330513954, |
| "num_tokens": 1642996.0, |
| "reward": 0.7444743514060974, |
| "reward_std": 0.3576819598674774, |
| "rewards/correctness/mean": 0.8333333134651184, |
| "rewards/correctness/std": 0.3758230209350586, |
| "rewards/length_penalty/mean": -0.08885905146598816, |
| "rewards/length_penalty/std": 0.05079879239201546, |
| "sampling/importance_sampling_ratio/max": 1.4455554485321045, |
| "sampling/importance_sampling_ratio/mean": 0.9900694489479065, |
| "sampling/importance_sampling_ratio/min": 0.6753557920455933, |
| "sampling/sampling_logp_difference/max": 0.3925156593322754, |
| "sampling/sampling_logp_difference/mean": 0.018528403714299202, |
| "step": 69, |
| "step_time": 8.946724934270605 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007477621402358636, |
| "clip_ratio/high_mean": 0.0007477621402358636, |
| "clip_ratio/low_mean": 8.721437188796699e-05, |
| "clip_ratio/low_min": 8.721437188796699e-05, |
| "clip_ratio/region_mean": 0.0008349765218251074, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 753.0, |
| "completions/max_terminated_length": 753.0, |
| "completions/mean_length": 209.6333465576172, |
| "completions/mean_terminated_length": 209.6333465576172, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.2410158837834994, |
| "epoch": 0.18229166666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007661667186766863, |
| "learning_rate": 1e-05, |
| "loss": 0.004670525901019573, |
| "num_tokens": 1660144.0, |
| "reward": 0.3143066465854645, |
| "reward_std": 0.529586672782898, |
| "rewards/correctness/mean": 0.4166666567325592, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.10236002504825592, |
| "rewards/length_penalty/std": 0.06358043849468231, |
| "sampling/importance_sampling_ratio/max": 1.533227801322937, |
| "sampling/importance_sampling_ratio/mean": 0.9915273785591125, |
| "sampling/importance_sampling_ratio/min": 0.6519303321838379, |
| "sampling/sampling_logp_difference/max": 0.42781758308410645, |
| "sampling/sampling_logp_difference/mean": 0.01636781543493271, |
| "step": 70, |
| "step_time": 9.379186490084976 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006305074978930255, |
| "clip_ratio/high_mean": 0.0006305074978930255, |
| "clip_ratio/low_mean": 8.671523149435718e-05, |
| "clip_ratio/low_min": 8.671523149435718e-05, |
| "clip_ratio/region_mean": 0.0007172227293873826, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 349.0, |
| "completions/max_terminated_length": 349.0, |
| "completions/mean_length": 173.85000610351562, |
| "completions/mean_terminated_length": 173.85000610351562, |
| "completions/min_length": 106.0, |
| "completions/min_terminated_length": 106.0, |
| "entropy": 0.21683228015899658, |
| "epoch": 0.18489583333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005390848498791456, |
| "learning_rate": 1e-05, |
| "loss": -5.334336310625076e-05, |
| "num_tokens": 1674755.0, |
| "reward": 0.49844565987586975, |
| "reward_std": 0.49669742584228516, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.08488769829273224, |
| "rewards/length_penalty/std": 0.020763332024216652, |
| "sampling/importance_sampling_ratio/max": 1.4671045541763306, |
| "sampling/importance_sampling_ratio/mean": 0.9921537637710571, |
| "sampling/importance_sampling_ratio/min": 0.5582857728004456, |
| "sampling/sampling_logp_difference/max": 0.5828843116760254, |
| "sampling/sampling_logp_difference/mean": 0.014949919655919075, |
| "step": 71, |
| "step_time": 4.9017251851037145 |
| }, |
| { |
| "clip_ratio/high_max": 0.001236946719776218, |
| "clip_ratio/high_mean": 0.001236946719776218, |
| "clip_ratio/low_mean": 0.0002472799193734924, |
| "clip_ratio/low_min": 0.0002472799193734924, |
| "clip_ratio/region_mean": 0.0014842266391497105, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 444.0, |
| "completions/max_terminated_length": 444.0, |
| "completions/mean_length": 163.1333465576172, |
| "completions/mean_terminated_length": 163.1333465576172, |
| "completions/min_length": 45.0, |
| "completions/min_terminated_length": 45.0, |
| "entropy": 0.2774425546328227, |
| "epoch": 0.1875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006716904696077108, |
| "learning_rate": 1e-05, |
| "loss": 0.0044620707631111145, |
| "num_tokens": 1688383.0, |
| "reward": 0.8870117664337158, |
| "reward_std": 0.19346240162849426, |
| "rewards/correctness/mean": 0.9666666388511658, |
| "rewards/correctness/std": 0.1810203343629837, |
| "rewards/length_penalty/mean": -0.07965494692325592, |
| "rewards/length_penalty/std": 0.035877082496881485, |
| "sampling/importance_sampling_ratio/max": 1.3777556419372559, |
| "sampling/importance_sampling_ratio/mean": 0.9898649454116821, |
| "sampling/importance_sampling_ratio/min": 0.637740969657898, |
| "sampling/sampling_logp_difference/max": 0.44982314109802246, |
| "sampling/sampling_logp_difference/mean": 0.01773042231798172, |
| "step": 72, |
| "step_time": 6.339901325991377 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009163227902414898, |
| "clip_ratio/high_mean": 0.0009163227902414898, |
| "clip_ratio/low_mean": 0.00022767439804738387, |
| "clip_ratio/low_min": 0.00022767439804738387, |
| "clip_ratio/region_mean": 0.001143997166461001, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1790.0, |
| "completions/mean_length": 653.6500244140625, |
| "completions/mean_terminated_length": 605.5689697265625, |
| "completions/min_length": 162.0, |
| "completions/min_terminated_length": 162.0, |
| "entropy": 0.3177306254704793, |
| "epoch": 0.19010416666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00891516450792551, |
| "learning_rate": 1e-05, |
| "loss": -0.006704499013721943, |
| "num_tokens": 1733142.0, |
| "reward": 0.16416829824447632, |
| "reward_std": 0.6266658306121826, |
| "rewards/correctness/mean": 0.4833333194255829, |
| "rewards/correctness/std": 0.5039392709732056, |
| "rewards/length_penalty/mean": -0.31916505098342896, |
| "rewards/length_penalty/std": 0.22116518020629883, |
| "sampling/importance_sampling_ratio/max": 1.6468907594680786, |
| "sampling/importance_sampling_ratio/mean": 0.9887369275093079, |
| "sampling/importance_sampling_ratio/min": 0.6219922304153442, |
| "sampling/sampling_logp_difference/max": 0.49888908863067627, |
| "sampling/sampling_logp_difference/mean": 0.018961848691105843, |
| "step": 73, |
| "step_time": 25.08421320770867 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008805921922127405, |
| "clip_ratio/high_mean": 0.0008805921922127405, |
| "clip_ratio/low_mean": 5.378078882737706e-05, |
| "clip_ratio/low_min": 5.378078882737706e-05, |
| "clip_ratio/region_mean": 0.0009343729810401177, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1932.0, |
| "completions/max_terminated_length": 1932.0, |
| "completions/mean_length": 336.7166748046875, |
| "completions/mean_terminated_length": 336.7166748046875, |
| "completions/min_length": 97.0, |
| "completions/min_terminated_length": 97.0, |
| "entropy": 0.23614277690649033, |
| "epoch": 0.19270833333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008565094321966171, |
| "learning_rate": 1e-05, |
| "loss": -0.017027905210852623, |
| "num_tokens": 1757945.0, |
| "reward": 0.40225425362586975, |
| "reward_std": 0.4861847162246704, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.16441243886947632, |
| "rewards/length_penalty/std": 0.1560681164264679, |
| "sampling/importance_sampling_ratio/max": 1.5113457441329956, |
| "sampling/importance_sampling_ratio/mean": 0.9917812943458557, |
| "sampling/importance_sampling_ratio/min": 0.47178271412849426, |
| "sampling/sampling_logp_difference/max": 0.7512367963790894, |
| "sampling/sampling_logp_difference/mean": 0.01505094114691019, |
| "step": 74, |
| "step_time": 21.79212696524337 |
| }, |
| { |
| "clip_ratio/high_max": 0.001553242977630968, |
| "clip_ratio/high_mean": 0.001553242977630968, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001553242977630968, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 546.0, |
| "completions/max_terminated_length": 546.0, |
| "completions/mean_length": 167.0500030517578, |
| "completions/mean_terminated_length": 167.0500030517578, |
| "completions/min_length": 68.0, |
| "completions/min_terminated_length": 68.0, |
| "entropy": 0.2676567832628886, |
| "epoch": 0.1953125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0071274456568062305, |
| "learning_rate": 1e-05, |
| "loss": 0.0031374837271869183, |
| "num_tokens": 1773488.0, |
| "reward": 0.6350992918014526, |
| "reward_std": 0.47557058930397034, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.4544196128845215, |
| "rewards/length_penalty/mean": -0.08156738430261612, |
| "rewards/length_penalty/std": 0.04113834351301193, |
| "sampling/importance_sampling_ratio/max": 1.3971836566925049, |
| "sampling/importance_sampling_ratio/mean": 0.9901525378227234, |
| "sampling/importance_sampling_ratio/min": 0.6605772972106934, |
| "sampling/sampling_logp_difference/max": 0.4146411418914795, |
| "sampling/sampling_logp_difference/mean": 0.01780414581298828, |
| "step": 75, |
| "step_time": 7.203926707152277 |
| }, |
| { |
| "clip_ratio/high_max": 0.00041760620176016044, |
| "clip_ratio/high_mean": 0.00041760620176016044, |
| "clip_ratio/low_mean": 0.00017063397778353343, |
| "clip_ratio/low_min": 0.00017063397778353343, |
| "clip_ratio/region_mean": 0.0005882401795436939, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1287.0, |
| "completions/max_terminated_length": 1287.0, |
| "completions/mean_length": 242.35000610351562, |
| "completions/mean_terminated_length": 242.35000610351562, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.2644062489271164, |
| "epoch": 0.19791666666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009264216758310795, |
| "learning_rate": 1e-05, |
| "loss": 0.02456854283809662, |
| "num_tokens": 1792559.0, |
| "reward": 0.5983317494392395, |
| "reward_std": 0.508339524269104, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.4544196128845215, |
| "rewards/length_penalty/mean": -0.11833496391773224, |
| "rewards/length_penalty/std": 0.10933779180049896, |
| "sampling/importance_sampling_ratio/max": 1.3920600414276123, |
| "sampling/importance_sampling_ratio/mean": 0.9900718927383423, |
| "sampling/importance_sampling_ratio/min": 0.6488889455795288, |
| "sampling/sampling_logp_difference/max": 0.4324936866760254, |
| "sampling/sampling_logp_difference/mean": 0.01711069978773594, |
| "step": 76, |
| "step_time": 15.279150035930797 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010006486506123717, |
| "clip_ratio/high_mean": 0.0010006486506123717, |
| "clip_ratio/low_mean": 8.477450076801081e-05, |
| "clip_ratio/low_min": 8.477450076801081e-05, |
| "clip_ratio/region_mean": 0.0010854231416791056, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 599.0, |
| "completions/mean_length": 281.70001220703125, |
| "completions/mean_terminated_length": 251.76271057128906, |
| "completions/min_length": 45.0, |
| "completions/min_terminated_length": 45.0, |
| "entropy": 0.25790657103061676, |
| "epoch": 0.20052083333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00790585856884718, |
| "learning_rate": 1e-05, |
| "loss": 0.020197078585624695, |
| "num_tokens": 1815831.0, |
| "reward": 0.262451171875, |
| "reward_std": 0.5297642946243286, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.13754883408546448, |
| "rewards/length_penalty/std": 0.1298300176858902, |
| "sampling/importance_sampling_ratio/max": 1.4403636455535889, |
| "sampling/importance_sampling_ratio/mean": 0.9906535744667053, |
| "sampling/importance_sampling_ratio/min": 0.6435641050338745, |
| "sampling/sampling_logp_difference/max": 0.4407336711883545, |
| "sampling/sampling_logp_difference/mean": 0.01626359112560749, |
| "step": 77, |
| "step_time": 22.936317761195824 |
| }, |
| { |
| "clip_ratio/high_max": 0.001179996373442312, |
| "clip_ratio/high_mean": 0.001179996373442312, |
| "clip_ratio/low_mean": 0.0002230996712266157, |
| "clip_ratio/low_min": 0.0002230996712266157, |
| "clip_ratio/region_mean": 0.0014030960446689278, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1619.0, |
| "completions/mean_length": 348.4666748046875, |
| "completions/mean_terminated_length": 319.6610107421875, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.38759085536003113, |
| "epoch": 0.203125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009313313290476799, |
| "learning_rate": 1e-05, |
| "loss": -0.0005689286626875401, |
| "num_tokens": 1840979.0, |
| "reward": 0.5298503041267395, |
| "reward_std": 0.6099588871002197, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.17014974355697632, |
| "rewards/length_penalty/std": 0.18531392514705658, |
| "sampling/importance_sampling_ratio/max": 1.4484256505966187, |
| "sampling/importance_sampling_ratio/mean": 0.9857113361358643, |
| "sampling/importance_sampling_ratio/min": 0.6241329312324524, |
| "sampling/sampling_logp_difference/max": 0.4713919162750244, |
| "sampling/sampling_logp_difference/mean": 0.02441319264471531, |
| "step": 78, |
| "step_time": 23.314662725664675 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016396614179636042, |
| "clip_ratio/high_mean": 0.0016396614179636042, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0016396614179636042, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 840.0, |
| "completions/max_terminated_length": 840.0, |
| "completions/mean_length": 237.11668395996094, |
| "completions/mean_terminated_length": 237.11668395996094, |
| "completions/min_length": 42.0, |
| "completions/min_terminated_length": 42.0, |
| "entropy": 0.2744365856051445, |
| "epoch": 0.20572916666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0061734686605632305, |
| "learning_rate": 1e-05, |
| "loss": -0.005295232869684696, |
| "num_tokens": 1860296.0, |
| "reward": 0.7008870840072632, |
| "reward_std": 0.4385649263858795, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.11577962338924408, |
| "rewards/length_penalty/std": 0.08675486594438553, |
| "sampling/importance_sampling_ratio/max": 1.3452285528182983, |
| "sampling/importance_sampling_ratio/mean": 0.9904361963272095, |
| "sampling/importance_sampling_ratio/min": 0.6681658625602722, |
| "sampling/sampling_logp_difference/max": 0.4032188653945923, |
| "sampling/sampling_logp_difference/mean": 0.017866255715489388, |
| "step": 79, |
| "step_time": 10.31341484002769 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009620860995103916, |
| "clip_ratio/high_mean": 0.0009620860995103916, |
| "clip_ratio/low_mean": 5.340168718248606e-05, |
| "clip_ratio/low_min": 5.340168718248606e-05, |
| "clip_ratio/region_mean": 0.0010154877866928775, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 861.0, |
| "completions/max_terminated_length": 861.0, |
| "completions/mean_length": 224.933349609375, |
| "completions/mean_terminated_length": 224.933349609375, |
| "completions/min_length": 27.0, |
| "completions/min_terminated_length": 27.0, |
| "entropy": 0.35508089264233905, |
| "epoch": 0.20833333333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008378067053854465, |
| "learning_rate": 1e-05, |
| "loss": -0.0017847889102995396, |
| "num_tokens": 1877302.0, |
| "reward": 0.5401692986488342, |
| "reward_std": 0.4942512810230255, |
| "rewards/correctness/mean": 0.6499999761581421, |
| "rewards/correctness/std": 0.48099473118782043, |
| "rewards/length_penalty/mean": -0.10983072966337204, |
| "rewards/length_penalty/std": 0.09769939631223679, |
| "sampling/importance_sampling_ratio/max": 1.4081530570983887, |
| "sampling/importance_sampling_ratio/mean": 0.9869513511657715, |
| "sampling/importance_sampling_ratio/min": 0.6328408122062683, |
| "sampling/sampling_logp_difference/max": 0.4575364589691162, |
| "sampling/sampling_logp_difference/mean": 0.022432325407862663, |
| "step": 80, |
| "step_time": 10.159734084969386 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010435700678499416, |
| "clip_ratio/high_mean": 0.0010435700678499416, |
| "clip_ratio/low_mean": 0.0001097213050040106, |
| "clip_ratio/low_min": 0.0001097213050040106, |
| "clip_ratio/region_mean": 0.0011532913922565058, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1355.0, |
| "completions/max_terminated_length": 1355.0, |
| "completions/mean_length": 222.85000610351562, |
| "completions/mean_terminated_length": 222.85000610351562, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.26323076834281284, |
| "epoch": 0.2109375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008192833513021469, |
| "learning_rate": 1e-05, |
| "loss": -0.023273896425962448, |
| "num_tokens": 1895783.0, |
| "reward": 0.5078532099723816, |
| "reward_std": 0.502953827381134, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014302253723, |
| "rewards/length_penalty/mean": -0.10881347954273224, |
| "rewards/length_penalty/std": 0.09745525568723679, |
| "sampling/importance_sampling_ratio/max": 1.3580296039581299, |
| "sampling/importance_sampling_ratio/mean": 0.9909989237785339, |
| "sampling/importance_sampling_ratio/min": 0.7111003994941711, |
| "sampling/sampling_logp_difference/max": 0.3409416675567627, |
| "sampling/sampling_logp_difference/mean": 0.016630185768008232, |
| "step": 81, |
| "step_time": 15.49428718117997 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005951060156803578, |
| "clip_ratio/high_mean": 0.0005951060156803578, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0005951060156803578, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 328.0, |
| "completions/max_terminated_length": 328.0, |
| "completions/mean_length": 150.73333740234375, |
| "completions/mean_terminated_length": 150.73333740234375, |
| "completions/min_length": 52.0, |
| "completions/min_terminated_length": 52.0, |
| "entropy": 0.2501143490274747, |
| "epoch": 0.21354166666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005451194941997528, |
| "learning_rate": 1e-05, |
| "loss": 0.015339905396103859, |
| "num_tokens": 1908237.0, |
| "reward": 0.5930664539337158, |
| "reward_std": 0.47544047236442566, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.07360026240348816, |
| "rewards/length_penalty/std": 0.02788763865828514, |
| "sampling/importance_sampling_ratio/max": 1.2841066122055054, |
| "sampling/importance_sampling_ratio/mean": 0.9911366701126099, |
| "sampling/importance_sampling_ratio/min": 0.6299075484275818, |
| "sampling/sampling_logp_difference/max": 0.4621821641921997, |
| "sampling/sampling_logp_difference/mean": 0.01694982871413231, |
| "step": 82, |
| "step_time": 5.103488169144839 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008281492143093298, |
| "clip_ratio/high_mean": 0.0008281492143093298, |
| "clip_ratio/low_mean": 0.00013143466154967123, |
| "clip_ratio/low_min": 0.00013143466154967123, |
| "clip_ratio/region_mean": 0.0009595838782843202, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1387.0, |
| "completions/max_terminated_length": 1387.0, |
| "completions/mean_length": 302.3500061035156, |
| "completions/mean_terminated_length": 302.3500061035156, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.32328422367572784, |
| "epoch": 0.21614583333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007696948014199734, |
| "learning_rate": 1e-05, |
| "loss": -0.00271498691290617, |
| "num_tokens": 1930528.0, |
| "reward": 0.41903483867645264, |
| "reward_std": 0.5690031051635742, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.14763183891773224, |
| "rewards/length_penalty/std": 0.13010774552822113, |
| "sampling/importance_sampling_ratio/max": 1.397360920906067, |
| "sampling/importance_sampling_ratio/mean": 0.9889217615127563, |
| "sampling/importance_sampling_ratio/min": 0.13567645847797394, |
| "sampling/sampling_logp_difference/max": 1.9974822998046875, |
| "sampling/sampling_logp_difference/mean": 0.020394140854477882, |
| "step": 83, |
| "step_time": 15.862845247378573 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011307440387705963, |
| "clip_ratio/high_mean": 0.0011307440387705963, |
| "clip_ratio/low_mean": 0.0001332952087977901, |
| "clip_ratio/low_min": 0.0001332952087977901, |
| "clip_ratio/region_mean": 0.0012640392475683864, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 756.0, |
| "completions/max_terminated_length": 756.0, |
| "completions/mean_length": 216.73333740234375, |
| "completions/mean_terminated_length": 216.73333740234375, |
| "completions/min_length": 54.0, |
| "completions/min_terminated_length": 54.0, |
| "entropy": 0.2779706120491028, |
| "epoch": 0.21875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007516487035900354, |
| "learning_rate": 1e-05, |
| "loss": -0.001617394620552659, |
| "num_tokens": 1949392.0, |
| "reward": 0.3775065243244171, |
| "reward_std": 0.5321787595748901, |
| "rewards/correctness/mean": 0.4833333194255829, |
| "rewards/correctness/std": 0.5039392709732056, |
| "rewards/length_penalty/mean": -0.10582682490348816, |
| "rewards/length_penalty/std": 0.08100845664739609, |
| "sampling/importance_sampling_ratio/max": 1.4396148920059204, |
| "sampling/importance_sampling_ratio/mean": 0.9900568127632141, |
| "sampling/importance_sampling_ratio/min": 0.6550421714782715, |
| "sampling/sampling_logp_difference/max": 0.42305564880371094, |
| "sampling/sampling_logp_difference/mean": 0.01845000870525837, |
| "step": 84, |
| "step_time": 9.552567144157365 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014517546466474112, |
| "clip_ratio/high_mean": 0.0014517546466474112, |
| "clip_ratio/low_mean": 8.701948293795188e-05, |
| "clip_ratio/low_min": 8.701948293795188e-05, |
| "clip_ratio/region_mean": 0.0015387741247347246, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1761.0, |
| "completions/mean_length": 313.6166687011719, |
| "completions/mean_terminated_length": 284.2203369140625, |
| "completions/min_length": 72.0, |
| "completions/min_terminated_length": 72.0, |
| "entropy": 0.3463781674702962, |
| "epoch": 0.22135416666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0054717715829610825, |
| "learning_rate": 1e-05, |
| "loss": -0.009404251351952553, |
| "num_tokens": 1971819.0, |
| "reward": 0.5968669056892395, |
| "reward_std": 0.5676841735839844, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.15313313901424408, |
| "rewards/length_penalty/std": 0.19131147861480713, |
| "sampling/importance_sampling_ratio/max": 1.5409660339355469, |
| "sampling/importance_sampling_ratio/mean": 0.987484335899353, |
| "sampling/importance_sampling_ratio/min": 0.6595447659492493, |
| "sampling/sampling_logp_difference/max": 0.43240952491760254, |
| "sampling/sampling_logp_difference/mean": 0.02106129191815853, |
| "step": 85, |
| "step_time": 23.029872465413064 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006091700003404791, |
| "clip_ratio/high_mean": 0.0006091700003404791, |
| "clip_ratio/low_mean": 0.00012496430038784942, |
| "clip_ratio/low_min": 0.00012496430038784942, |
| "clip_ratio/region_mean": 0.0007341342958776901, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1375.0, |
| "completions/mean_length": 441.5500183105469, |
| "completions/mean_terminated_length": 263.0555725097656, |
| "completions/min_length": 124.0, |
| "completions/min_terminated_length": 124.0, |
| "entropy": 0.2880704899628957, |
| "epoch": 0.22395833333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009985826909542084, |
| "learning_rate": 1e-05, |
| "loss": 0.09145986288785934, |
| "num_tokens": 2002432.0, |
| "reward": 0.5010660886764526, |
| "reward_std": 0.6522772312164307, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.45441964268684387, |
| "rewards/length_penalty/mean": -0.21560057997703552, |
| "rewards/length_penalty/std": 0.2804999053478241, |
| "sampling/importance_sampling_ratio/max": 1.373631477355957, |
| "sampling/importance_sampling_ratio/mean": 0.9892457723617554, |
| "sampling/importance_sampling_ratio/min": 0.6765832901000977, |
| "sampling/sampling_logp_difference/max": 0.39069974422454834, |
| "sampling/sampling_logp_difference/mean": 0.01817106269299984, |
| "step": 86, |
| "step_time": 23.85519070387818 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006988735743410265, |
| "clip_ratio/high_mean": 0.0006988735743410265, |
| "clip_ratio/low_mean": 0.0002471166565859069, |
| "clip_ratio/low_min": 0.0002471166565859069, |
| "clip_ratio/region_mean": 0.0009459902309269334, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1807.0, |
| "completions/mean_length": 427.3000183105469, |
| "completions/mean_terminated_length": 342.0, |
| "completions/min_length": 59.0, |
| "completions/min_terminated_length": 59.0, |
| "entropy": 0.2381190707286199, |
| "epoch": 0.2265625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010000785812735558, |
| "learning_rate": 1e-05, |
| "loss": 0.01194899994879961, |
| "num_tokens": 2031700.0, |
| "reward": 0.19135743379592896, |
| "reward_std": 0.6304640173912048, |
| "rewards/correctness/mean": 0.4000000059604645, |
| "rewards/correctness/std": 0.49403220415115356, |
| "rewards/length_penalty/mean": -0.20864257216453552, |
| "rewards/length_penalty/std": 0.26017582416534424, |
| "sampling/importance_sampling_ratio/max": 1.3940961360931396, |
| "sampling/importance_sampling_ratio/mean": 0.9916189312934875, |
| "sampling/importance_sampling_ratio/min": 0.4005557596683502, |
| "sampling/sampling_logp_difference/max": 0.9149023294448853, |
| "sampling/sampling_logp_difference/mean": 0.015000944957137108, |
| "step": 87, |
| "step_time": 24.1907924041152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011908001421640317, |
| "clip_ratio/high_mean": 0.0011908001421640317, |
| "clip_ratio/low_mean": 0.000395987454491357, |
| "clip_ratio/low_min": 0.000395987454491357, |
| "clip_ratio/region_mean": 0.0015867875966553886, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 890.0, |
| "completions/max_terminated_length": 890.0, |
| "completions/mean_length": 237.7166748046875, |
| "completions/mean_terminated_length": 237.7166748046875, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.2755401283502579, |
| "epoch": 0.22916666666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007770989090204239, |
| "learning_rate": 1e-05, |
| "loss": 0.0006232140585780144, |
| "num_tokens": 2050363.0, |
| "reward": 0.5339274406433105, |
| "reward_std": 0.49097785353660583, |
| "rewards/correctness/mean": 0.6499999761581421, |
| "rewards/correctness/std": 0.48099473118782043, |
| "rewards/length_penalty/mean": -0.11607258766889572, |
| "rewards/length_penalty/std": 0.08006425201892853, |
| "sampling/importance_sampling_ratio/max": 1.4670826196670532, |
| "sampling/importance_sampling_ratio/mean": 0.9900525808334351, |
| "sampling/importance_sampling_ratio/min": 0.5982344746589661, |
| "sampling/sampling_logp_difference/max": 0.5137724876403809, |
| "sampling/sampling_logp_difference/mean": 0.01778515614569187, |
| "step": 88, |
| "step_time": 10.635103869950399 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018204191680221509, |
| "clip_ratio/high_mean": 0.0018204191680221509, |
| "clip_ratio/low_mean": 0.000392135310297211, |
| "clip_ratio/low_min": 0.000392135310297211, |
| "clip_ratio/region_mean": 0.0022125544492155313, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1395.0, |
| "completions/mean_length": 323.7833557128906, |
| "completions/mean_terminated_length": 294.559326171875, |
| "completions/min_length": 99.0, |
| "completions/min_terminated_length": 99.0, |
| "entropy": 0.4288439104954402, |
| "epoch": 0.23177083333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010807260870933533, |
| "learning_rate": 1e-05, |
| "loss": 0.020352285355329514, |
| "num_tokens": 2076410.0, |
| "reward": 0.4085693657398224, |
| "reward_std": 0.5347954630851746, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.15809732675552368, |
| "rewards/length_penalty/std": 0.1692076027393341, |
| "sampling/importance_sampling_ratio/max": 1.4190025329589844, |
| "sampling/importance_sampling_ratio/mean": 0.9853951334953308, |
| "sampling/importance_sampling_ratio/min": 0.6728789210319519, |
| "sampling/sampling_logp_difference/max": 0.39618992805480957, |
| "sampling/sampling_logp_difference/mean": 0.02459992654621601, |
| "step": 89, |
| "step_time": 24.274444866226986 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008052853130114576, |
| "clip_ratio/high_mean": 0.0008052853130114576, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008052853130114576, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1997.0, |
| "completions/mean_length": 387.2333679199219, |
| "completions/mean_terminated_length": 359.0847473144531, |
| "completions/min_length": 99.0, |
| "completions/min_terminated_length": 99.0, |
| "entropy": 0.21026979883511862, |
| "epoch": 0.234375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006514720153063536, |
| "learning_rate": 1e-05, |
| "loss": 0.0008430758025497198, |
| "num_tokens": 2103374.0, |
| "reward": 0.1109212264418602, |
| "reward_std": 0.5491532683372498, |
| "rewards/correctness/mean": 0.30000001192092896, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.18907877802848816, |
| "rewards/length_penalty/std": 0.19972538948059082, |
| "sampling/importance_sampling_ratio/max": 1.3757601976394653, |
| "sampling/importance_sampling_ratio/mean": 0.9923238158226013, |
| "sampling/importance_sampling_ratio/min": 0.4560941755771637, |
| "sampling/sampling_logp_difference/max": 0.7850559949874878, |
| "sampling/sampling_logp_difference/mean": 0.01392011996358633, |
| "step": 90, |
| "step_time": 23.65313331480138 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008332563253740469, |
| "clip_ratio/high_mean": 0.0008332563253740469, |
| "clip_ratio/low_mean": 7.997440601078172e-05, |
| "clip_ratio/low_min": 7.997440601078172e-05, |
| "clip_ratio/region_mean": 0.0009132307313848287, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 329.0, |
| "completions/max_terminated_length": 329.0, |
| "completions/mean_length": 185.06668090820312, |
| "completions/mean_terminated_length": 185.06668090820312, |
| "completions/min_length": 112.0, |
| "completions/min_terminated_length": 112.0, |
| "entropy": 0.2601093426346779, |
| "epoch": 0.23697916666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006987820379436016, |
| "learning_rate": 1e-05, |
| "loss": -0.0019227066077291965, |
| "num_tokens": 2118378.0, |
| "reward": 0.592968761920929, |
| "reward_std": 0.4775463938713074, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.09036458283662796, |
| "rewards/length_penalty/std": 0.02380477637052536, |
| "sampling/importance_sampling_ratio/max": 1.4670376777648926, |
| "sampling/importance_sampling_ratio/mean": 0.9903293251991272, |
| "sampling/importance_sampling_ratio/min": 0.6078482866287231, |
| "sampling/sampling_logp_difference/max": 0.4978299140930176, |
| "sampling/sampling_logp_difference/mean": 0.0169296283274889, |
| "step": 91, |
| "step_time": 5.205353772034869 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005360775394365191, |
| "clip_ratio/high_mean": 0.0005360775394365191, |
| "clip_ratio/low_mean": 0.0002713084880573054, |
| "clip_ratio/low_min": 0.0002713084880573054, |
| "clip_ratio/region_mean": 0.0008073860177925477, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1602.0, |
| "completions/max_terminated_length": 1602.0, |
| "completions/mean_length": 314.5333557128906, |
| "completions/mean_terminated_length": 314.5333557128906, |
| "completions/min_length": 92.0, |
| "completions/min_terminated_length": 92.0, |
| "entropy": 0.3107937425374985, |
| "epoch": 0.23958333333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00717154610902071, |
| "learning_rate": 1e-05, |
| "loss": 0.010096949525177479, |
| "num_tokens": 2140880.0, |
| "reward": 0.3964192867279053, |
| "reward_std": 0.5628794431686401, |
| "rewards/correctness/mean": 0.550000011920929, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.15358072519302368, |
| "rewards/length_penalty/std": 0.1413908153772354, |
| "sampling/importance_sampling_ratio/max": 1.4221454858779907, |
| "sampling/importance_sampling_ratio/mean": 0.9887853860855103, |
| "sampling/importance_sampling_ratio/min": 0.6308075785636902, |
| "sampling/sampling_logp_difference/max": 0.46075439453125, |
| "sampling/sampling_logp_difference/mean": 0.019710902124643326, |
| "step": 92, |
| "step_time": 18.346877998672426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012314619671087712, |
| "clip_ratio/high_mean": 0.0012314619671087712, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0012314619671087712, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1467.0, |
| "completions/max_terminated_length": 1467.0, |
| "completions/mean_length": 261.70001220703125, |
| "completions/mean_terminated_length": 261.70001220703125, |
| "completions/min_length": 54.0, |
| "completions/min_terminated_length": 54.0, |
| "entropy": 0.27854517102241516, |
| "epoch": 0.2421875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008893297053873539, |
| "learning_rate": 1e-05, |
| "loss": 0.00923291314393282, |
| "num_tokens": 2160112.0, |
| "reward": 0.6888834834098816, |
| "reward_std": 0.4526672661304474, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.12778320908546448, |
| "rewards/length_penalty/std": 0.1341627836227417, |
| "sampling/importance_sampling_ratio/max": 1.6626660823822021, |
| "sampling/importance_sampling_ratio/mean": 0.9899617433547974, |
| "sampling/importance_sampling_ratio/min": 0.6380678415298462, |
| "sampling/sampling_logp_difference/max": 0.5084223747253418, |
| "sampling/sampling_logp_difference/mean": 0.01806235872209072, |
| "step": 93, |
| "step_time": 16.523078211117536 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008830397418932989, |
| "clip_ratio/high_mean": 0.0008830397418932989, |
| "clip_ratio/low_mean": 9.504799769880871e-05, |
| "clip_ratio/low_min": 9.504799769880871e-05, |
| "clip_ratio/region_mean": 0.000978087744442746, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1686.0, |
| "completions/mean_length": 286.5500183105469, |
| "completions/mean_terminated_length": 256.6949157714844, |
| "completions/min_length": 24.0, |
| "completions/min_terminated_length": 24.0, |
| "entropy": 0.37756167848904926, |
| "epoch": 0.24479166666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006421149242669344, |
| "learning_rate": 1e-05, |
| "loss": 0.0016049710102379322, |
| "num_tokens": 2182115.0, |
| "reward": 0.49341636896133423, |
| "reward_std": 0.6001912355422974, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.13991698622703552, |
| "rewards/length_penalty/std": 0.1931096613407135, |
| "sampling/importance_sampling_ratio/max": 1.4276680946350098, |
| "sampling/importance_sampling_ratio/mean": 0.9863218665122986, |
| "sampling/importance_sampling_ratio/min": 0.6635614633560181, |
| "sampling/sampling_logp_difference/max": 0.41013383865356445, |
| "sampling/sampling_logp_difference/mean": 0.022641684859991074, |
| "step": 94, |
| "step_time": 23.30174124869518 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010107319491604965, |
| "clip_ratio/high_mean": 0.0010107319491604965, |
| "clip_ratio/low_mean": 4.9236827180720866e-05, |
| "clip_ratio/low_min": 4.9236827180720866e-05, |
| "clip_ratio/region_mean": 0.0010599687811918557, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1180.0, |
| "completions/max_terminated_length": 1180.0, |
| "completions/mean_length": 246.60000610351562, |
| "completions/mean_terminated_length": 246.60000610351562, |
| "completions/min_length": 49.0, |
| "completions/min_terminated_length": 49.0, |
| "entropy": 0.2527260233958562, |
| "epoch": 0.24739583333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008922259323298931, |
| "learning_rate": 1e-05, |
| "loss": 0.019261308014392853, |
| "num_tokens": 2200931.0, |
| "reward": 0.47958987951278687, |
| "reward_std": 0.5174242854118347, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.12041015923023224, |
| "rewards/length_penalty/std": 0.10505987703800201, |
| "sampling/importance_sampling_ratio/max": 1.7255852222442627, |
| "sampling/importance_sampling_ratio/mean": 0.9909000396728516, |
| "sampling/importance_sampling_ratio/min": 0.6726726293563843, |
| "sampling/sampling_logp_difference/max": 0.545566201210022, |
| "sampling/sampling_logp_difference/mean": 0.01613456755876541, |
| "step": 95, |
| "step_time": 13.464065812760964 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006967556740467747, |
| "clip_ratio/high_mean": 0.0006967556740467747, |
| "clip_ratio/low_mean": 0.0002910077746491879, |
| "clip_ratio/low_min": 0.0002910077746491879, |
| "clip_ratio/region_mean": 0.0009877634583972394, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2034.0, |
| "completions/mean_length": 561.800048828125, |
| "completions/mean_terminated_length": 536.6101684570312, |
| "completions/min_length": 64.0, |
| "completions/min_terminated_length": 64.0, |
| "entropy": 0.3483681281407674, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011937614530324936, |
| "learning_rate": 1e-05, |
| "loss": 0.017384085804224014, |
| "num_tokens": 2239039.0, |
| "reward": 0.29235026240348816, |
| "reward_std": 0.6839615702629089, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.2743164002895355, |
| "rewards/length_penalty/std": 0.2696426808834076, |
| "sampling/importance_sampling_ratio/max": 1.760434865951538, |
| "sampling/importance_sampling_ratio/mean": 0.987684428691864, |
| "sampling/importance_sampling_ratio/min": 0.5473218560218811, |
| "sampling/sampling_logp_difference/max": 0.6027182340621948, |
| "sampling/sampling_logp_difference/mean": 0.021040918305516243, |
| "step": 96, |
| "step_time": 24.926648694789037 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003127544478047639, |
| "clip_ratio/high_mean": 0.0003127544478047639, |
| "clip_ratio/low_mean": 0.00031005687681802857, |
| "clip_ratio/low_min": 0.00031005687681802857, |
| "clip_ratio/region_mean": 0.0006228113294734309, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1970.0, |
| "completions/max_terminated_length": 1970.0, |
| "completions/mean_length": 350.38336181640625, |
| "completions/mean_terminated_length": 350.38336181640625, |
| "completions/min_length": 68.0, |
| "completions/min_terminated_length": 68.0, |
| "entropy": 0.3482793718576431, |
| "epoch": 0.2526041666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009928246960043907, |
| "learning_rate": 1e-05, |
| "loss": 0.004756946116685867, |
| "num_tokens": 2263522.0, |
| "reward": 0.39558106660842896, |
| "reward_std": 0.6178069114685059, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.17108561098575592, |
| "rewards/length_penalty/std": 0.198683962225914, |
| "sampling/importance_sampling_ratio/max": 1.4684332609176636, |
| "sampling/importance_sampling_ratio/mean": 0.9876013994216919, |
| "sampling/importance_sampling_ratio/min": 0.6131176948547363, |
| "sampling/sampling_logp_difference/max": 0.4891984462738037, |
| "sampling/sampling_logp_difference/mean": 0.020787904039025307, |
| "step": 97, |
| "step_time": 22.525413759052753 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007667019478200624, |
| "clip_ratio/high_mean": 0.0007667019478200624, |
| "clip_ratio/low_mean": 0.00020109389151912183, |
| "clip_ratio/low_min": 0.00020109389151912183, |
| "clip_ratio/region_mean": 0.0009677958538910995, |
| "completions/clipped_ratio": 0.06666667014360428, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1914.0, |
| "completions/mean_length": 493.3000183105469, |
| "completions/mean_terminated_length": 382.2500305175781, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.34876831372578937, |
| "epoch": 0.2552083333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012217363342642784, |
| "learning_rate": 1e-05, |
| "loss": 0.0969504565000534, |
| "num_tokens": 2297190.0, |
| "reward": 0.5591309070587158, |
| "reward_std": 0.6514180302619934, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4033755958080292, |
| "rewards/length_penalty/mean": -0.24086913466453552, |
| "rewards/length_penalty/std": 0.282613605260849, |
| "sampling/importance_sampling_ratio/max": 1.6478209495544434, |
| "sampling/importance_sampling_ratio/mean": 0.9880919456481934, |
| "sampling/importance_sampling_ratio/min": 0.5854965448379517, |
| "sampling/sampling_logp_difference/max": 0.5352950096130371, |
| "sampling/sampling_logp_difference/mean": 0.020098721608519554, |
| "step": 98, |
| "step_time": 24.131559785688296 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004372035352086338, |
| "clip_ratio/high_mean": 0.0004372035352086338, |
| "clip_ratio/low_mean": 0.0002675797392536576, |
| "clip_ratio/low_min": 0.0002675797392536576, |
| "clip_ratio/region_mean": 0.000704783281738249, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1636.0, |
| "completions/mean_length": 492.36669921875, |
| "completions/mean_terminated_length": 410.4912414550781, |
| "completions/min_length": 79.0, |
| "completions/min_terminated_length": 79.0, |
| "entropy": 0.3205321654677391, |
| "epoch": 0.2578125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012764493934810162, |
| "learning_rate": 1e-05, |
| "loss": 0.017153235152363777, |
| "num_tokens": 2330492.0, |
| "reward": 0.3262532651424408, |
| "reward_std": 0.694017767906189, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.24041341245174408, |
| "rewards/length_penalty/std": 0.27765506505966187, |
| "sampling/importance_sampling_ratio/max": 1.719120979309082, |
| "sampling/importance_sampling_ratio/mean": 0.9883583784103394, |
| "sampling/importance_sampling_ratio/min": 0.5768380761146545, |
| "sampling/sampling_logp_difference/max": 0.5501937866210938, |
| "sampling/sampling_logp_difference/mean": 0.019611768424510956, |
| "step": 99, |
| "step_time": 24.04859541170299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005960500129731372, |
| "clip_ratio/high_mean": 0.0005960500129731372, |
| "clip_ratio/low_mean": 0.00025546454223028076, |
| "clip_ratio/low_min": 0.00025546454223028076, |
| "clip_ratio/region_mean": 0.0008515145455021411, |
| "completions/clipped_ratio": 0.28333336114883423, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1780.0, |
| "completions/mean_length": 877.0833740234375, |
| "completions/mean_terminated_length": 414.16278076171875, |
| "completions/min_length": 67.0, |
| "completions/min_terminated_length": 67.0, |
| "entropy": 0.3811613420645396, |
| "epoch": 0.2604166666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010149510577321053, |
| "learning_rate": 1e-05, |
| "loss": 0.02623521164059639, |
| "num_tokens": 2388907.0, |
| "reward": 0.0717366561293602, |
| "reward_std": 0.835002601146698, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5042194724082947, |
| "rewards/length_penalty/mean": -0.4282633364200592, |
| "rewards/length_penalty/std": 0.40259334444999695, |
| "sampling/importance_sampling_ratio/max": 1.5463850498199463, |
| "sampling/importance_sampling_ratio/mean": 0.9867529273033142, |
| "sampling/importance_sampling_ratio/min": 0.32857051491737366, |
| "sampling/sampling_logp_difference/max": 1.1130037307739258, |
| "sampling/sampling_logp_difference/mean": 0.022446660324931145, |
| "step": 100, |
| "step_time": 26.83455875958316 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011162592951829235, |
| "clip_ratio/high_mean": 0.0011162592951829235, |
| "clip_ratio/low_mean": 0.00019129625676820675, |
| "clip_ratio/low_min": 0.00019129625676820675, |
| "clip_ratio/region_mean": 0.0013075555810549606, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1918.0, |
| "completions/mean_length": 735.8500366210938, |
| "completions/mean_terminated_length": 590.0555419921875, |
| "completions/min_length": 51.0, |
| "completions/min_terminated_length": 51.0, |
| "entropy": 0.5312706977128983, |
| "epoch": 0.2630208333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009944680146872997, |
| "learning_rate": 1e-05, |
| "loss": 0.012254921719431877, |
| "num_tokens": 2437578.0, |
| "reward": 0.07403157651424408, |
| "reward_std": 0.7417707443237305, |
| "rewards/correctness/mean": 0.4333333373069763, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.35930174589157104, |
| "rewards/length_penalty/std": 0.32801246643066406, |
| "sampling/importance_sampling_ratio/max": 1.5150156021118164, |
| "sampling/importance_sampling_ratio/mean": 0.9823935031890869, |
| "sampling/importance_sampling_ratio/min": 0.45981356501579285, |
| "sampling/sampling_logp_difference/max": 0.7769341468811035, |
| "sampling/sampling_logp_difference/mean": 0.02888464741408825, |
| "step": 101, |
| "step_time": 25.413738016737625 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009321634230824808, |
| "clip_ratio/high_mean": 0.0009321634230824808, |
| "clip_ratio/low_mean": 0.00016390664192537466, |
| "clip_ratio/low_min": 0.00016390664192537466, |
| "clip_ratio/region_mean": 0.0010960700553065787, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 619.0, |
| "completions/max_terminated_length": 619.0, |
| "completions/mean_length": 224.25001525878906, |
| "completions/mean_terminated_length": 224.25001525878906, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.24270586917797723, |
| "epoch": 0.265625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007639285642653704, |
| "learning_rate": 1e-05, |
| "loss": -0.0022843158803880215, |
| "num_tokens": 2455213.0, |
| "reward": 0.5238363146781921, |
| "reward_std": 0.4734676778316498, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.1094970703125, |
| "rewards/length_penalty/std": 0.08380536735057831, |
| "sampling/importance_sampling_ratio/max": 1.437448263168335, |
| "sampling/importance_sampling_ratio/mean": 0.9917762875556946, |
| "sampling/importance_sampling_ratio/min": 0.6488960385322571, |
| "sampling/sampling_logp_difference/max": 0.4324827194213867, |
| "sampling/sampling_logp_difference/mean": 0.016455713659524918, |
| "step": 102, |
| "step_time": 8.081311426823959 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008517921378370374, |
| "clip_ratio/high_mean": 0.0008517921378370374, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008517921378370374, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 855.0, |
| "completions/max_terminated_length": 855.0, |
| "completions/mean_length": 295.7666931152344, |
| "completions/mean_terminated_length": 295.7666931152344, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.25733499228954315, |
| "epoch": 0.2682291666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006912329699844122, |
| "learning_rate": 1e-05, |
| "loss": 0.0071053411811590195, |
| "num_tokens": 2477079.0, |
| "reward": 0.33891603350639343, |
| "reward_std": 0.5632880330085754, |
| "rewards/correctness/mean": 0.4833333194255829, |
| "rewards/correctness/std": 0.5039392709732056, |
| "rewards/length_penalty/mean": -0.14441731572151184, |
| "rewards/length_penalty/std": 0.09639912843704224, |
| "sampling/importance_sampling_ratio/max": 1.4099655151367188, |
| "sampling/importance_sampling_ratio/mean": 0.9911353588104248, |
| "sampling/importance_sampling_ratio/min": 0.6556710600852966, |
| "sampling/sampling_logp_difference/max": 0.42209601402282715, |
| "sampling/sampling_logp_difference/mean": 0.016643350943922997, |
| "step": 103, |
| "step_time": 10.835010477574542 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013250188931124285, |
| "clip_ratio/high_mean": 0.0013250188931124285, |
| "clip_ratio/low_mean": 0.00026277409051544964, |
| "clip_ratio/low_min": 0.00026277409051544964, |
| "clip_ratio/region_mean": 0.001587792969075963, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1581.0, |
| "completions/mean_length": 445.70001220703125, |
| "completions/mean_terminated_length": 390.4482727050781, |
| "completions/min_length": 86.0, |
| "completions/min_terminated_length": 86.0, |
| "entropy": 0.3427726129690806, |
| "epoch": 0.2708333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010129323229193687, |
| "learning_rate": 1e-05, |
| "loss": 0.039104338735342026, |
| "num_tokens": 2507441.0, |
| "reward": 0.31570640206336975, |
| "reward_std": 0.5873957872390747, |
| "rewards/correctness/mean": 0.5333333611488342, |
| "rewards/correctness/std": 0.5030977725982666, |
| "rewards/length_penalty/mean": -0.21762695908546448, |
| "rewards/length_penalty/std": 0.23287583887577057, |
| "sampling/importance_sampling_ratio/max": 1.4507180452346802, |
| "sampling/importance_sampling_ratio/mean": 0.9869900345802307, |
| "sampling/importance_sampling_ratio/min": 0.626615047454834, |
| "sampling/sampling_logp_difference/max": 0.46742284297943115, |
| "sampling/sampling_logp_difference/mean": 0.021725626662373543, |
| "step": 104, |
| "step_time": 24.270891547203064 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013204141772196938, |
| "clip_ratio/high_mean": 0.0013204141772196938, |
| "clip_ratio/low_mean": 5.617346323560923e-05, |
| "clip_ratio/low_min": 5.617346323560923e-05, |
| "clip_ratio/region_mean": 0.0013765876453059416, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1509.0, |
| "completions/mean_length": 368.3000183105469, |
| "completions/mean_terminated_length": 339.83050537109375, |
| "completions/min_length": 95.0, |
| "completions/min_terminated_length": 95.0, |
| "entropy": 0.33968063195546466, |
| "epoch": 0.2734375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009448972530663013, |
| "learning_rate": 1e-05, |
| "loss": 0.01361564639955759, |
| "num_tokens": 2534499.0, |
| "reward": 0.3201660215854645, |
| "reward_std": 0.5672650933265686, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5042194724082947, |
| "rewards/length_penalty/mean": -0.17983397841453552, |
| "rewards/length_penalty/std": 0.17428958415985107, |
| "sampling/importance_sampling_ratio/max": 1.4172089099884033, |
| "sampling/importance_sampling_ratio/mean": 0.9881472587585449, |
| "sampling/importance_sampling_ratio/min": 0.6486535668373108, |
| "sampling/sampling_logp_difference/max": 0.4328565001487732, |
| "sampling/sampling_logp_difference/mean": 0.021434752270579338, |
| "step": 105, |
| "step_time": 23.66842356743291 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003586134795720379, |
| "clip_ratio/high_mean": 0.0003586134795720379, |
| "clip_ratio/low_mean": 9.735202183946967e-05, |
| "clip_ratio/low_min": 9.735202183946967e-05, |
| "clip_ratio/region_mean": 0.00045596550141150755, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 950.0, |
| "completions/max_terminated_length": 950.0, |
| "completions/mean_length": 191.50001525878906, |
| "completions/mean_terminated_length": 191.50001525878906, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.23151462276776633, |
| "epoch": 0.2760416666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005184860900044441, |
| "learning_rate": 1e-05, |
| "loss": -0.0035153753124177456, |
| "num_tokens": 2549559.0, |
| "reward": 0.8231608271598816, |
| "reward_std": 0.28046324849128723, |
| "rewards/correctness/mean": 0.9166666865348816, |
| "rewards/correctness/std": 0.2787178158760071, |
| "rewards/length_penalty/mean": -0.093505859375, |
| "rewards/length_penalty/std": 0.05774654820561409, |
| "sampling/importance_sampling_ratio/max": 1.3465787172317505, |
| "sampling/importance_sampling_ratio/mean": 0.9919623136520386, |
| "sampling/importance_sampling_ratio/min": 0.672252357006073, |
| "sampling/sampling_logp_difference/max": 0.3971214294433594, |
| "sampling/sampling_logp_difference/mean": 0.01552337035536766, |
| "step": 106, |
| "step_time": 10.755532223032787 |
| }, |
| { |
| "clip_ratio/high_max": 0.000989813250877584, |
| "clip_ratio/high_mean": 0.000989813250877584, |
| "clip_ratio/low_mean": 0.0004978063104014533, |
| "clip_ratio/low_min": 0.0004978063104014533, |
| "clip_ratio/region_mean": 0.0014876195637043566, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 970.0, |
| "completions/mean_length": 439.9833679199219, |
| "completions/mean_terminated_length": 261.3148193359375, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.4321523557106654, |
| "epoch": 0.2786458333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007592189125716686, |
| "learning_rate": 1e-05, |
| "loss": 0.007870635949075222, |
| "num_tokens": 2581008.0, |
| "reward": 0.20183107256889343, |
| "reward_std": 0.6510159969329834, |
| "rewards/correctness/mean": 0.4166666567325592, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.21483561396598816, |
| "rewards/length_penalty/std": 0.27571362257003784, |
| "sampling/importance_sampling_ratio/max": 1.396535038948059, |
| "sampling/importance_sampling_ratio/mean": 0.9849995374679565, |
| "sampling/importance_sampling_ratio/min": 0.38096120953559875, |
| "sampling/sampling_logp_difference/max": 0.9650577306747437, |
| "sampling/sampling_logp_difference/mean": 0.024866413325071335, |
| "step": 107, |
| "step_time": 24.275926110567525 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010699949052650481, |
| "clip_ratio/high_mean": 0.0010699949052650481, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0010699949052650481, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 309.0, |
| "completions/max_terminated_length": 309.0, |
| "completions/mean_length": 166.83334350585938, |
| "completions/mean_terminated_length": 166.83334350585938, |
| "completions/min_length": 64.0, |
| "completions/min_terminated_length": 64.0, |
| "entropy": 0.23725493748982748, |
| "epoch": 0.28125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005343505181372166, |
| "learning_rate": 1e-05, |
| "loss": 0.0021302206441760063, |
| "num_tokens": 2594388.0, |
| "reward": 0.40187177062034607, |
| "reward_std": 0.5278782844543457, |
| "rewards/correctness/mean": 0.4833333194255829, |
| "rewards/correctness/std": 0.5039392709732056, |
| "rewards/length_penalty/mean": -0.0814615860581398, |
| "rewards/length_penalty/std": 0.02909676358103752, |
| "sampling/importance_sampling_ratio/max": 1.3279615640640259, |
| "sampling/importance_sampling_ratio/mean": 0.9913597702980042, |
| "sampling/importance_sampling_ratio/min": 0.5493717193603516, |
| "sampling/sampling_logp_difference/max": 0.5989799499511719, |
| "sampling/sampling_logp_difference/mean": 0.016212156042456627, |
| "step": 108, |
| "step_time": 4.563385086366907 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008405640061634282, |
| "clip_ratio/high_mean": 0.0008405640061634282, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0008405640061634282, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 250.0, |
| "completions/max_terminated_length": 250.0, |
| "completions/mean_length": 145.5166778564453, |
| "completions/mean_terminated_length": 145.5166778564453, |
| "completions/min_length": 61.0, |
| "completions/min_terminated_length": 61.0, |
| "entropy": 0.2610517392555873, |
| "epoch": 0.2838541666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005832832306623459, |
| "learning_rate": 1e-05, |
| "loss": 0.001013047294691205, |
| "num_tokens": 2606809.0, |
| "reward": 0.7956136465072632, |
| "reward_std": 0.34949249029159546, |
| "rewards/correctness/mean": 0.8666666746139526, |
| "rewards/correctness/std": 0.34280332922935486, |
| "rewards/length_penalty/mean": -0.07105305790901184, |
| "rewards/length_penalty/std": 0.025378605350852013, |
| "sampling/importance_sampling_ratio/max": 1.319395661354065, |
| "sampling/importance_sampling_ratio/mean": 0.9910116791725159, |
| "sampling/importance_sampling_ratio/min": 0.6801387667655945, |
| "sampling/sampling_logp_difference/max": 0.38545846939086914, |
| "sampling/sampling_logp_difference/mean": 0.017436077818274498, |
| "step": 109, |
| "step_time": 3.984763785963878 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011856376319580402, |
| "clip_ratio/high_mean": 0.0011856376319580402, |
| "clip_ratio/low_mean": 0.00016015999426599592, |
| "clip_ratio/low_min": 0.00016015999426599592, |
| "clip_ratio/region_mean": 0.001345797626224036, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1847.0, |
| "completions/max_terminated_length": 1847.0, |
| "completions/mean_length": 262.2666931152344, |
| "completions/mean_terminated_length": 262.2666931152344, |
| "completions/min_length": 110.0, |
| "completions/min_terminated_length": 110.0, |
| "entropy": 0.28981975466012955, |
| "epoch": 0.2864583333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009128313511610031, |
| "learning_rate": 1e-05, |
| "loss": 0.023787569254636765, |
| "num_tokens": 2626065.0, |
| "reward": 0.6886067986488342, |
| "reward_std": 0.43836256861686707, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.12805989384651184, |
| "rewards/length_penalty/std": 0.11576338112354279, |
| "sampling/importance_sampling_ratio/max": 1.3840749263763428, |
| "sampling/importance_sampling_ratio/mean": 0.9893162250518799, |
| "sampling/importance_sampling_ratio/min": 0.6533727645874023, |
| "sampling/sampling_logp_difference/max": 0.42560744285583496, |
| "sampling/sampling_logp_difference/mean": 0.018311815336346626, |
| "step": 110, |
| "step_time": 20.719824175350368 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015197760076262057, |
| "clip_ratio/high_mean": 0.0015197760076262057, |
| "clip_ratio/low_mean": 0.00026497703705293435, |
| "clip_ratio/low_min": 0.00026497703705293435, |
| "clip_ratio/region_mean": 0.0017847530155753095, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 983.0, |
| "completions/mean_length": 233.9166717529297, |
| "completions/mean_terminated_length": 203.16949462890625, |
| "completions/min_length": 82.0, |
| "completions/min_terminated_length": 82.0, |
| "entropy": 0.25897983213265735, |
| "epoch": 0.2890625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009503885172307491, |
| "learning_rate": 1e-05, |
| "loss": 0.05067107081413269, |
| "num_tokens": 2644130.0, |
| "reward": 0.769116222858429, |
| "reward_std": 0.40131640434265137, |
| "rewards/correctness/mean": 0.8833333253860474, |
| "rewards/correctness/std": 0.32373178005218506, |
| "rewards/length_penalty/mean": -0.1142171248793602, |
| "rewards/length_penalty/std": 0.13351398706436157, |
| "sampling/importance_sampling_ratio/max": 1.4320305585861206, |
| "sampling/importance_sampling_ratio/mean": 0.990073561668396, |
| "sampling/importance_sampling_ratio/min": 0.6474267840385437, |
| "sampling/sampling_logp_difference/max": 0.4347496032714844, |
| "sampling/sampling_logp_difference/mean": 0.017878778278827667, |
| "step": 111, |
| "step_time": 22.28089099866338 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011310035333735868, |
| "clip_ratio/high_mean": 0.0011310035333735868, |
| "clip_ratio/low_mean": 8.999538840726018e-05, |
| "clip_ratio/low_min": 8.999538840726018e-05, |
| "clip_ratio/region_mean": 0.0012209989023782934, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1195.0, |
| "completions/max_terminated_length": 1195.0, |
| "completions/mean_length": 319.66668701171875, |
| "completions/mean_terminated_length": 319.66668701171875, |
| "completions/min_length": 79.0, |
| "completions/min_terminated_length": 79.0, |
| "entropy": 0.2511361415187518, |
| "epoch": 0.2916666666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010103827342391014, |
| "learning_rate": 1e-05, |
| "loss": 0.018800456076860428, |
| "num_tokens": 2667710.0, |
| "reward": 0.6605794429779053, |
| "reward_std": 0.4402918219566345, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.1560872346162796, |
| "rewards/length_penalty/std": 0.11278058588504791, |
| "sampling/importance_sampling_ratio/max": 1.4245244264602661, |
| "sampling/importance_sampling_ratio/mean": 0.9910459518432617, |
| "sampling/importance_sampling_ratio/min": 0.43974727392196655, |
| "sampling/sampling_logp_difference/max": 0.8215551376342773, |
| "sampling/sampling_logp_difference/mean": 0.01640731655061245, |
| "step": 112, |
| "step_time": 13.939699458191171 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007322309053658197, |
| "clip_ratio/high_mean": 0.0007322309053658197, |
| "clip_ratio/low_mean": 0.0002101000888311925, |
| "clip_ratio/low_min": 0.0002101000888311925, |
| "clip_ratio/region_mean": 0.0009423310063236082, |
| "completions/clipped_ratio": 0.11666667461395264, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 723.0, |
| "completions/mean_length": 442.13336181640625, |
| "completions/mean_terminated_length": 230.03773498535156, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.36275026698907215, |
| "epoch": 0.2942708333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007664171513170004, |
| "learning_rate": 1e-05, |
| "loss": 0.0032901261001825333, |
| "num_tokens": 2699668.0, |
| "reward": 0.5174479484558105, |
| "reward_std": 0.6714236736297607, |
| "rewards/correctness/mean": 0.7333333492279053, |
| "rewards/correctness/std": 0.4459484815597534, |
| "rewards/length_penalty/mean": -0.21588541567325592, |
| "rewards/length_penalty/std": 0.290835976600647, |
| "sampling/importance_sampling_ratio/max": 1.409894585609436, |
| "sampling/importance_sampling_ratio/mean": 0.9867975115776062, |
| "sampling/importance_sampling_ratio/min": 0.49045974016189575, |
| "sampling/sampling_logp_difference/max": 0.7124121189117432, |
| "sampling/sampling_logp_difference/mean": 0.021570686250925064, |
| "step": 113, |
| "step_time": 24.13513475214131 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009021415802029272, |
| "clip_ratio/high_mean": 0.0009021415802029272, |
| "clip_ratio/low_mean": 0.00017731636762619019, |
| "clip_ratio/low_min": 0.00017731636762619019, |
| "clip_ratio/region_mean": 0.0010794579623810325, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 824.0, |
| "completions/max_terminated_length": 824.0, |
| "completions/mean_length": 268.63336181640625, |
| "completions/mean_terminated_length": 268.63336181640625, |
| "completions/min_length": 91.0, |
| "completions/min_terminated_length": 91.0, |
| "entropy": 0.3083602388699849, |
| "epoch": 0.296875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007035584654659033, |
| "learning_rate": 1e-05, |
| "loss": -0.00703608151525259, |
| "num_tokens": 2718776.0, |
| "reward": 0.3854980766773224, |
| "reward_std": 0.5536922216415405, |
| "rewards/correctness/mean": 0.5166666507720947, |
| "rewards/correctness/std": 0.5039393305778503, |
| "rewards/length_penalty/mean": -0.13116861879825592, |
| "rewards/length_penalty/std": 0.08917821943759918, |
| "sampling/importance_sampling_ratio/max": 1.6358071565628052, |
| "sampling/importance_sampling_ratio/mean": 0.9884124994277954, |
| "sampling/importance_sampling_ratio/min": 0.6594327092170715, |
| "sampling/sampling_logp_difference/max": 0.4921363592147827, |
| "sampling/sampling_logp_difference/mean": 0.019618535414338112, |
| "step": 114, |
| "step_time": 10.3174605127424 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015310987510019913, |
| "clip_ratio/high_mean": 0.0015310987510019913, |
| "clip_ratio/low_mean": 0.0001732217885243396, |
| "clip_ratio/low_min": 0.0001732217885243396, |
| "clip_ratio/region_mean": 0.0017043205201237772, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 473.0, |
| "completions/max_terminated_length": 473.0, |
| "completions/mean_length": 208.40000915527344, |
| "completions/mean_terminated_length": 208.40000915527344, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.22973039001226425, |
| "epoch": 0.2994791666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006087599787861109, |
| "learning_rate": 1e-05, |
| "loss": -0.0016181793762370944, |
| "num_tokens": 2734930.0, |
| "reward": 0.31490886211395264, |
| "reward_std": 0.5272872447967529, |
| "rewards/correctness/mean": 0.4166666567325592, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.10175780951976776, |
| "rewards/length_penalty/std": 0.055966559797525406, |
| "sampling/importance_sampling_ratio/max": 1.4407658576965332, |
| "sampling/importance_sampling_ratio/mean": 0.9916548728942871, |
| "sampling/importance_sampling_ratio/min": 0.6377993226051331, |
| "sampling/sampling_logp_difference/max": 0.44973158836364746, |
| "sampling/sampling_logp_difference/mean": 0.01533613819628954, |
| "step": 115, |
| "step_time": 6.253021043958142 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007972561191612234, |
| "clip_ratio/high_mean": 0.0007972561191612234, |
| "clip_ratio/low_mean": 0.00011618025018833578, |
| "clip_ratio/low_min": 0.00011618025018833578, |
| "clip_ratio/region_mean": 0.000913436379050836, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1272.0, |
| "completions/max_terminated_length": 1272.0, |
| "completions/mean_length": 244.86668395996094, |
| "completions/mean_terminated_length": 244.86668395996094, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.28110752006371814, |
| "epoch": 0.3020833333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012378980405628681, |
| "learning_rate": 1e-05, |
| "loss": 0.018241384997963905, |
| "num_tokens": 2755052.0, |
| "reward": 0.41376954317092896, |
| "reward_std": 0.4986647665500641, |
| "rewards/correctness/mean": 0.5333333611488342, |
| "rewards/correctness/std": 0.5030977725982666, |
| "rewards/length_penalty/mean": -0.11956380307674408, |
| "rewards/length_penalty/std": 0.10308168083429337, |
| "sampling/importance_sampling_ratio/max": 1.70135498046875, |
| "sampling/importance_sampling_ratio/mean": 0.9901013374328613, |
| "sampling/importance_sampling_ratio/min": 0.4468599855899811, |
| "sampling/sampling_logp_difference/max": 0.8055100440979004, |
| "sampling/sampling_logp_difference/mean": 0.01864585466682911, |
| "step": 116, |
| "step_time": 14.436019308166578 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008476423972751945, |
| "clip_ratio/high_mean": 0.0008476423972751945, |
| "clip_ratio/low_mean": 7.53806719634061e-05, |
| "clip_ratio/low_min": 7.53806719634061e-05, |
| "clip_ratio/region_mean": 0.000923023074089239, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 870.0, |
| "completions/max_terminated_length": 870.0, |
| "completions/mean_length": 182.2166748046875, |
| "completions/mean_terminated_length": 182.2166748046875, |
| "completions/min_length": 64.0, |
| "completions/min_terminated_length": 64.0, |
| "entropy": 0.360212321082751, |
| "epoch": 0.3046875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008254350163042545, |
| "learning_rate": 1e-05, |
| "loss": -0.0038145408034324646, |
| "num_tokens": 2769265.0, |
| "reward": 0.49436038732528687, |
| "reward_std": 0.5239414572715759, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.08897297829389572, |
| "rewards/length_penalty/std": 0.06797512620687485, |
| "sampling/importance_sampling_ratio/max": 1.4467955827713013, |
| "sampling/importance_sampling_ratio/mean": 0.9868465065956116, |
| "sampling/importance_sampling_ratio/min": 0.6463786363601685, |
| "sampling/sampling_logp_difference/max": 0.4363698959350586, |
| "sampling/sampling_logp_difference/mean": 0.02208094857633114, |
| "step": 117, |
| "step_time": 9.882572655100375 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007841558447883775, |
| "clip_ratio/high_mean": 0.0007841558447883775, |
| "clip_ratio/low_mean": 0.00016625367667681226, |
| "clip_ratio/low_min": 0.00016625367667681226, |
| "clip_ratio/region_mean": 0.0009504095166145513, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 438.0, |
| "completions/max_terminated_length": 438.0, |
| "completions/mean_length": 191.03334045410156, |
| "completions/mean_terminated_length": 191.03334045410156, |
| "completions/min_length": 53.0, |
| "completions/min_terminated_length": 53.0, |
| "entropy": 0.24461991836627325, |
| "epoch": 0.3072916666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007504720240831375, |
| "learning_rate": 1e-05, |
| "loss": -0.0005558086559176445, |
| "num_tokens": 2784577.0, |
| "reward": 0.5067220330238342, |
| "reward_std": 0.4970279932022095, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.09327799826860428, |
| "rewards/length_penalty/std": 0.04192177951335907, |
| "sampling/importance_sampling_ratio/max": 1.411384105682373, |
| "sampling/importance_sampling_ratio/mean": 0.991538405418396, |
| "sampling/importance_sampling_ratio/min": 0.6577974557876587, |
| "sampling/sampling_logp_difference/max": 0.41885828971862793, |
| "sampling/sampling_logp_difference/mean": 0.01619027368724346, |
| "step": 118, |
| "step_time": 5.80338279879652 |
| }, |
| { |
| "clip_ratio/high_max": 0.000646352828577316, |
| "clip_ratio/high_mean": 0.000646352828577316, |
| "clip_ratio/low_mean": 0.00023329600905223438, |
| "clip_ratio/low_min": 0.00023329600905223438, |
| "clip_ratio/region_mean": 0.0008796488255029544, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1153.0, |
| "completions/mean_length": 274.0333557128906, |
| "completions/mean_terminated_length": 243.96609497070312, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.33367759982744855, |
| "epoch": 0.3098958333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008914864622056484, |
| "learning_rate": 1e-05, |
| "loss": 0.02442135475575924, |
| "num_tokens": 2804909.0, |
| "reward": 0.4328613579273224, |
| "reward_std": 0.566518247127533, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.13380533456802368, |
| "rewards/length_penalty/std": 0.15720155835151672, |
| "sampling/importance_sampling_ratio/max": 1.6677967309951782, |
| "sampling/importance_sampling_ratio/mean": 0.9880881309509277, |
| "sampling/importance_sampling_ratio/min": 0.49594399333000183, |
| "sampling/sampling_logp_difference/max": 0.7012922763824463, |
| "sampling/sampling_logp_difference/mean": 0.020485220476984978, |
| "step": 119, |
| "step_time": 22.57717277179472 |
| }, |
| { |
| "clip_ratio/high_max": 0.0003936410406216358, |
| "clip_ratio/high_mean": 0.0003936410406216358, |
| "clip_ratio/low_mean": 0.00019690683499599496, |
| "clip_ratio/low_min": 0.00019690683499599496, |
| "clip_ratio/region_mean": 0.0005905478707669923, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1206.0, |
| "completions/mean_length": 327.433349609375, |
| "completions/mean_terminated_length": 268.10345458984375, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.32935722917318344, |
| "epoch": 0.3125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007574846968054771, |
| "learning_rate": 1e-05, |
| "loss": 0.022690845653414726, |
| "num_tokens": 2828845.0, |
| "reward": 0.10678711533546448, |
| "reward_std": 0.5209624171257019, |
| "rewards/correctness/mean": 0.2666666805744171, |
| "rewards/correctness/std": 0.4459485113620758, |
| "rewards/length_penalty/mean": -0.15987955033779144, |
| "rewards/length_penalty/std": 0.19083140790462494, |
| "sampling/importance_sampling_ratio/max": 1.4401273727416992, |
| "sampling/importance_sampling_ratio/mean": 0.9879485964775085, |
| "sampling/importance_sampling_ratio/min": 0.5835459232330322, |
| "sampling/sampling_logp_difference/max": 0.5386321544647217, |
| "sampling/sampling_logp_difference/mean": 0.020554877817630768, |
| "step": 120, |
| "step_time": 24.038608172209933 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005714220363491526, |
| "clip_ratio/high_mean": 0.0005714220363491526, |
| "clip_ratio/low_mean": 0.00010707732872106135, |
| "clip_ratio/low_min": 0.00010707732872106135, |
| "clip_ratio/region_mean": 0.0006784993650702139, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 779.0, |
| "completions/mean_length": 375.61669921875, |
| "completions/mean_terminated_length": 189.79629516601562, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.18538923437396684, |
| "epoch": 0.3151041666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007036050781607628, |
| "learning_rate": 1e-05, |
| "loss": 0.07639457285404205, |
| "num_tokens": 2855182.0, |
| "reward": 0.3165934383869171, |
| "reward_std": 0.6698235869407654, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5042194724082947, |
| "rewards/length_penalty/mean": -0.18340657651424408, |
| "rewards/length_penalty/std": 0.2798919677734375, |
| "sampling/importance_sampling_ratio/max": 1.4506728649139404, |
| "sampling/importance_sampling_ratio/mean": 0.993786633014679, |
| "sampling/importance_sampling_ratio/min": 0.5941579341888428, |
| "sampling/sampling_logp_difference/max": 0.5206100940704346, |
| "sampling/sampling_logp_difference/mean": 0.010517205111682415, |
| "step": 121, |
| "step_time": 23.484501062193885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007296871796521979, |
| "clip_ratio/high_mean": 0.0007296871796521979, |
| "clip_ratio/low_mean": 0.0002878263670330246, |
| "clip_ratio/low_min": 0.0002878263670330246, |
| "clip_ratio/region_mean": 0.0010175135491105418, |
| "completions/clipped_ratio": 0.0833333358168602, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1901.0, |
| "completions/mean_length": 507.4833679199219, |
| "completions/mean_terminated_length": 367.43634033203125, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.3637760082880656, |
| "epoch": 0.3177083333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012505095452070236, |
| "learning_rate": 1e-05, |
| "loss": 0.01897193305194378, |
| "num_tokens": 2890221.0, |
| "reward": 0.4522054195404053, |
| "reward_std": 0.6964542269706726, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.24779459834098816, |
| "rewards/length_penalty/std": 0.3065641522407532, |
| "sampling/importance_sampling_ratio/max": 1.6379073858261108, |
| "sampling/importance_sampling_ratio/mean": 0.9869912266731262, |
| "sampling/importance_sampling_ratio/min": 0.6758850812911987, |
| "sampling/sampling_logp_difference/max": 0.4934194087982178, |
| "sampling/sampling_logp_difference/mean": 0.022409377619624138, |
| "step": 122, |
| "step_time": 24.364205190213397 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011325166754735012, |
| "clip_ratio/high_mean": 0.0011325166754735012, |
| "clip_ratio/low_mean": 0.00015280135752012333, |
| "clip_ratio/low_min": 0.00015280135752012333, |
| "clip_ratio/region_mean": 0.0012853180693734128, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2011.0, |
| "completions/mean_length": 435.0166931152344, |
| "completions/mean_terminated_length": 379.39654541015625, |
| "completions/min_length": 62.0, |
| "completions/min_terminated_length": 62.0, |
| "entropy": 0.4102325240770976, |
| "epoch": 0.3203125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013061443343758583, |
| "learning_rate": 1e-05, |
| "loss": 0.005935294553637505, |
| "num_tokens": 2920162.0, |
| "reward": 0.4875895380973816, |
| "reward_std": 0.6004350781440735, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.21241047978401184, |
| "rewards/length_penalty/std": 0.25518375635147095, |
| "sampling/importance_sampling_ratio/max": 1.450933814048767, |
| "sampling/importance_sampling_ratio/mean": 0.9857973456382751, |
| "sampling/importance_sampling_ratio/min": 0.6262840628623962, |
| "sampling/sampling_logp_difference/max": 0.46795129776000977, |
| "sampling/sampling_logp_difference/mean": 0.02386726438999176, |
| "step": 123, |
| "step_time": 23.91245395829901 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016021693057458226, |
| "clip_ratio/high_mean": 0.0016021693057458226, |
| "clip_ratio/low_mean": 0.0001980951007377977, |
| "clip_ratio/low_min": 0.0001980951007377977, |
| "clip_ratio/region_mean": 0.0018002644113342587, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 576.0, |
| "completions/max_terminated_length": 576.0, |
| "completions/mean_length": 190.9166717529297, |
| "completions/mean_terminated_length": 190.9166717529297, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.24027026444673538, |
| "epoch": 0.3229166666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009001859463751316, |
| "learning_rate": 1e-05, |
| "loss": -0.0018330328166484833, |
| "num_tokens": 2935707.0, |
| "reward": 0.790112316608429, |
| "reward_std": 0.3424948453903198, |
| "rewards/correctness/mean": 0.8833333253860474, |
| "rewards/correctness/std": 0.32373178005218506, |
| "rewards/length_penalty/mean": -0.0932210311293602, |
| "rewards/length_penalty/std": 0.05391903221607208, |
| "sampling/importance_sampling_ratio/max": 1.330735206604004, |
| "sampling/importance_sampling_ratio/mean": 0.9910055994987488, |
| "sampling/importance_sampling_ratio/min": 0.636630117893219, |
| "sampling/sampling_logp_difference/max": 0.4515664577484131, |
| "sampling/sampling_logp_difference/mean": 0.01618552766740322, |
| "step": 124, |
| "step_time": 7.980881370138377 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007010912571179991, |
| "clip_ratio/high_mean": 0.0007010912571179991, |
| "clip_ratio/low_mean": 8.869966647277276e-05, |
| "clip_ratio/low_min": 8.869966647277276e-05, |
| "clip_ratio/region_mean": 0.0007897909041882182, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 386.0, |
| "completions/max_terminated_length": 386.0, |
| "completions/mean_length": 214.7666778564453, |
| "completions/mean_terminated_length": 214.7666778564453, |
| "completions/min_length": 118.0, |
| "completions/min_terminated_length": 118.0, |
| "entropy": 0.23325745264689127, |
| "epoch": 0.3255208333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0076285963878035545, |
| "learning_rate": 1e-05, |
| "loss": 0.0010199667885899544, |
| "num_tokens": 2952703.0, |
| "reward": 0.46180015802383423, |
| "reward_std": 0.5066788792610168, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.10486653447151184, |
| "rewards/length_penalty/std": 0.027354000136256218, |
| "sampling/importance_sampling_ratio/max": 1.4670041799545288, |
| "sampling/importance_sampling_ratio/mean": 0.9918311834335327, |
| "sampling/importance_sampling_ratio/min": 0.6016416549682617, |
| "sampling/sampling_logp_difference/max": 0.5080933570861816, |
| "sampling/sampling_logp_difference/mean": 0.015429210849106312, |
| "step": 125, |
| "step_time": 5.435763271059841 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005178170540602878, |
| "clip_ratio/high_mean": 0.0005178170540602878, |
| "clip_ratio/low_mean": 0.0002159235106470684, |
| "clip_ratio/low_min": 0.0002159235106470684, |
| "clip_ratio/region_mean": 0.0007337405598567178, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1428.0, |
| "completions/max_terminated_length": 1428.0, |
| "completions/mean_length": 358.3000183105469, |
| "completions/mean_terminated_length": 358.3000183105469, |
| "completions/min_length": 33.0, |
| "completions/min_terminated_length": 33.0, |
| "entropy": 0.2838987683256467, |
| "epoch": 0.328125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00928492657840252, |
| "learning_rate": 1e-05, |
| "loss": -0.005450396798551083, |
| "num_tokens": 2979861.0, |
| "reward": 0.1750488430261612, |
| "reward_std": 0.49175652861595154, |
| "rewards/correctness/mean": 0.3499999940395355, |
| "rewards/correctness/std": 0.48099473118782043, |
| "rewards/length_penalty/mean": -0.17495116591453552, |
| "rewards/length_penalty/std": 0.1544475257396698, |
| "sampling/importance_sampling_ratio/max": 1.4499104022979736, |
| "sampling/importance_sampling_ratio/mean": 0.9906268119812012, |
| "sampling/importance_sampling_ratio/min": 0.5180612206459045, |
| "sampling/sampling_logp_difference/max": 0.6576618552207947, |
| "sampling/sampling_logp_difference/mean": 0.017609527334570885, |
| "step": 126, |
| "step_time": 16.98496867576614 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007243898338250195, |
| "clip_ratio/high_mean": 0.0007243898338250195, |
| "clip_ratio/low_mean": 0.00021060557628516108, |
| "clip_ratio/low_min": 0.00021060557628516108, |
| "clip_ratio/region_mean": 0.0009349953955582654, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1211.0, |
| "completions/max_terminated_length": 1211.0, |
| "completions/mean_length": 378.5666809082031, |
| "completions/mean_terminated_length": 378.5666809082031, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.3279208143552144, |
| "epoch": 0.3307291666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00570105854421854, |
| "learning_rate": 1e-05, |
| "loss": -0.0011444264091551304, |
| "num_tokens": 3006075.0, |
| "reward": 0.2818196713924408, |
| "reward_std": 0.5993487238883972, |
| "rewards/correctness/mean": 0.46666666865348816, |
| "rewards/correctness/std": 0.5030977725982666, |
| "rewards/length_penalty/mean": -0.18484701216220856, |
| "rewards/length_penalty/std": 0.15096838772296906, |
| "sampling/importance_sampling_ratio/max": 1.4195140600204468, |
| "sampling/importance_sampling_ratio/mean": 0.9878858327865601, |
| "sampling/importance_sampling_ratio/min": 0.6423804759979248, |
| "sampling/sampling_logp_difference/max": 0.44257450103759766, |
| "sampling/sampling_logp_difference/mean": 0.020608557388186455, |
| "step": 127, |
| "step_time": 14.482137208338827 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018535105918999761, |
| "clip_ratio/high_mean": 0.0018535105918999761, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0018535105918999761, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 715.0, |
| "completions/max_terminated_length": 715.0, |
| "completions/mean_length": 259.41668701171875, |
| "completions/mean_terminated_length": 259.41668701171875, |
| "completions/min_length": 46.0, |
| "completions/min_terminated_length": 46.0, |
| "entropy": 0.31929664810498554, |
| "epoch": 0.3333333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007850701920688152, |
| "learning_rate": 1e-05, |
| "loss": -0.01267259195446968, |
| "num_tokens": 3026810.0, |
| "reward": 0.42333173751831055, |
| "reward_std": 0.5102311372756958, |
| "rewards/correctness/mean": 0.550000011920929, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.1266682893037796, |
| "rewards/length_penalty/std": 0.07224120199680328, |
| "sampling/importance_sampling_ratio/max": 1.3967772722244263, |
| "sampling/importance_sampling_ratio/mean": 0.9884933233261108, |
| "sampling/importance_sampling_ratio/min": 0.5949069857597351, |
| "sampling/sampling_logp_difference/max": 0.5193502902984619, |
| "sampling/sampling_logp_difference/mean": 0.020016726106405258, |
| "step": 128, |
| "step_time": 9.009117325069383 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018133548049566646, |
| "clip_ratio/high_mean": 0.0018133548049566646, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0018133548049566646, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 339.0, |
| "completions/max_terminated_length": 339.0, |
| "completions/mean_length": 159.60000610351562, |
| "completions/mean_terminated_length": 159.60000610351562, |
| "completions/min_length": 76.0, |
| "completions/min_terminated_length": 76.0, |
| "entropy": 0.27247823774814606, |
| "epoch": 0.3359375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00848185084760189, |
| "learning_rate": 1e-05, |
| "loss": 0.004966307431459427, |
| "num_tokens": 3040686.0, |
| "reward": 0.7054036855697632, |
| "reward_std": 0.4308811128139496, |
| "rewards/correctness/mean": 0.7833333611488342, |
| "rewards/correctness/std": 0.41545024514198303, |
| "rewards/length_penalty/mean": -0.07792969048023224, |
| "rewards/length_penalty/std": 0.027234794571995735, |
| "sampling/importance_sampling_ratio/max": 1.6072794198989868, |
| "sampling/importance_sampling_ratio/mean": 0.9901498556137085, |
| "sampling/importance_sampling_ratio/min": 0.657528817653656, |
| "sampling/sampling_logp_difference/max": 0.4745429754257202, |
| "sampling/sampling_logp_difference/mean": 0.01805119961500168, |
| "step": 129, |
| "step_time": 4.810239513404667 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007955170973824958, |
| "clip_ratio/high_mean": 0.0007955170973824958, |
| "clip_ratio/low_mean": 0.00016843523674954972, |
| "clip_ratio/low_min": 0.00016843523674954972, |
| "clip_ratio/region_mean": 0.0009639523535345992, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 735.0, |
| "completions/mean_length": 295.66668701171875, |
| "completions/mean_terminated_length": 235.2413787841797, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.26230694353580475, |
| "epoch": 0.3385416666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006255085114389658, |
| "learning_rate": 1e-05, |
| "loss": 0.0026140734553337097, |
| "num_tokens": 3062986.0, |
| "reward": 0.2889648675918579, |
| "reward_std": 0.5587278604507446, |
| "rewards/correctness/mean": 0.4333333373069763, |
| "rewards/correctness/std": 0.49971747398376465, |
| "rewards/length_penalty/mean": -0.1443684846162796, |
| "rewards/length_penalty/std": 0.1717526763677597, |
| "sampling/importance_sampling_ratio/max": 1.4325917959213257, |
| "sampling/importance_sampling_ratio/mean": 0.9905449748039246, |
| "sampling/importance_sampling_ratio/min": 0.6500624418258667, |
| "sampling/sampling_logp_difference/max": 0.4306868314743042, |
| "sampling/sampling_logp_difference/mean": 0.017099734395742416, |
| "step": 130, |
| "step_time": 23.814905456034467 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009188046242343262, |
| "clip_ratio/high_mean": 0.0009188046242343262, |
| "clip_ratio/low_mean": 0.00035014585591852665, |
| "clip_ratio/low_min": 0.00035014585591852665, |
| "clip_ratio/region_mean": 0.001268950494704768, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 951.0, |
| "completions/max_terminated_length": 951.0, |
| "completions/mean_length": 295.5333557128906, |
| "completions/mean_terminated_length": 295.5333557128906, |
| "completions/min_length": 27.0, |
| "completions/min_terminated_length": 27.0, |
| "entropy": 0.2733738124370575, |
| "epoch": 0.3411458333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0099636884406209, |
| "learning_rate": 1e-05, |
| "loss": -0.002763531170785427, |
| "num_tokens": 3085758.0, |
| "reward": 0.47236329317092896, |
| "reward_std": 0.5579666495323181, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.14430338144302368, |
| "rewards/length_penalty/std": 0.097950778901577, |
| "sampling/importance_sampling_ratio/max": 1.562045931816101, |
| "sampling/importance_sampling_ratio/mean": 0.9904019832611084, |
| "sampling/importance_sampling_ratio/min": 0.6248288154602051, |
| "sampling/sampling_logp_difference/max": 0.4702775478363037, |
| "sampling/sampling_logp_difference/mean": 0.017908092588186264, |
| "step": 131, |
| "step_time": 11.814676928101107 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008945932640926912, |
| "clip_ratio/high_mean": 0.0008945932640926912, |
| "clip_ratio/low_mean": 9.843421867117286e-05, |
| "clip_ratio/low_min": 9.843421867117286e-05, |
| "clip_ratio/region_mean": 0.0009930274876145024, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1575.0, |
| "completions/max_terminated_length": 1575.0, |
| "completions/mean_length": 231.15000915527344, |
| "completions/mean_terminated_length": 231.15000915527344, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.30564743528763455, |
| "epoch": 0.34375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0077720643021166325, |
| "learning_rate": 1e-05, |
| "loss": 0.023582246154546738, |
| "num_tokens": 3102677.0, |
| "reward": 0.40380048751831055, |
| "reward_std": 0.538102388381958, |
| "rewards/correctness/mean": 0.5166666507720947, |
| "rewards/correctness/std": 0.5039393305778503, |
| "rewards/length_penalty/mean": -0.11286620795726776, |
| "rewards/length_penalty/std": 0.10664499551057816, |
| "sampling/importance_sampling_ratio/max": 1.5083907842636108, |
| "sampling/importance_sampling_ratio/mean": 0.9882377982139587, |
| "sampling/importance_sampling_ratio/min": 0.6374589204788208, |
| "sampling/sampling_logp_difference/max": 0.45026540756225586, |
| "sampling/sampling_logp_difference/mean": 0.020378511399030685, |
| "step": 132, |
| "step_time": 17.14250988443382 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007735789549769834, |
| "clip_ratio/high_mean": 0.0007735789549769834, |
| "clip_ratio/low_mean": 0.00013787039400388798, |
| "clip_ratio/low_min": 0.00013787039400388798, |
| "clip_ratio/region_mean": 0.0009114493441302329, |
| "completions/clipped_ratio": 0.11666667461395264, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1034.0, |
| "completions/mean_length": 438.4833679199219, |
| "completions/mean_terminated_length": 225.90567016601562, |
| "completions/min_length": 91.0, |
| "completions/min_terminated_length": 91.0, |
| "entropy": 0.244797353943189, |
| "epoch": 0.3463541666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007750274147838354, |
| "learning_rate": 1e-05, |
| "loss": 0.018776727840304375, |
| "num_tokens": 3132266.0, |
| "reward": 0.4858968257904053, |
| "reward_std": 0.682120144367218, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.21410319209098816, |
| "rewards/length_penalty/std": 0.3013951778411865, |
| "sampling/importance_sampling_ratio/max": 1.6490556001663208, |
| "sampling/importance_sampling_ratio/mean": 0.9921458959579468, |
| "sampling/importance_sampling_ratio/min": 0.5881800651550293, |
| "sampling/sampling_logp_difference/max": 0.5307221412658691, |
| "sampling/sampling_logp_difference/mean": 0.013803096488118172, |
| "step": 133, |
| "step_time": 23.881334531120956 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008574419140738124, |
| "clip_ratio/high_mean": 0.0008574419140738124, |
| "clip_ratio/low_mean": 0.00015666137430040786, |
| "clip_ratio/low_min": 0.00015666137430040786, |
| "clip_ratio/region_mean": 0.0010141032786729436, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 463.0, |
| "completions/max_terminated_length": 463.0, |
| "completions/mean_length": 190.2666778564453, |
| "completions/mean_terminated_length": 190.2666778564453, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.2837923814853032, |
| "epoch": 0.3489583333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008241965435445309, |
| "learning_rate": 1e-05, |
| "loss": 0.0002308972179889679, |
| "num_tokens": 3147972.0, |
| "reward": 0.5237630605697632, |
| "reward_std": 0.5091028809547424, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.09290364384651184, |
| "rewards/length_penalty/std": 0.045004818588495255, |
| "sampling/importance_sampling_ratio/max": 1.3762222528457642, |
| "sampling/importance_sampling_ratio/mean": 0.9900805354118347, |
| "sampling/importance_sampling_ratio/min": 0.6296572089195251, |
| "sampling/sampling_logp_difference/max": 0.46257972717285156, |
| "sampling/sampling_logp_difference/mean": 0.018418744206428528, |
| "step": 134, |
| "step_time": 6.168864286504686 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006118701567174867, |
| "clip_ratio/high_mean": 0.0006118701567174867, |
| "clip_ratio/low_mean": 0.0001227910009523233, |
| "clip_ratio/low_min": 0.0001227910009523233, |
| "clip_ratio/region_mean": 0.0007346611528191715, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1771.0, |
| "completions/mean_length": 523.4666748046875, |
| "completions/mean_terminated_length": 354.0740661621094, |
| "completions/min_length": 77.0, |
| "completions/min_terminated_length": 77.0, |
| "entropy": 0.3021310120820999, |
| "epoch": 0.3515625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008444431237876415, |
| "learning_rate": 1e-05, |
| "loss": -0.019860858097672462, |
| "num_tokens": 3185220.0, |
| "reward": 0.4277344048023224, |
| "reward_std": 0.6905081272125244, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.2555989623069763, |
| "rewards/length_penalty/std": 0.29411980509757996, |
| "sampling/importance_sampling_ratio/max": 1.612852692604065, |
| "sampling/importance_sampling_ratio/mean": 0.9891533851623535, |
| "sampling/importance_sampling_ratio/min": 0.5679482221603394, |
| "sampling/sampling_logp_difference/max": 0.5657250881195068, |
| "sampling/sampling_logp_difference/mean": 0.01840287633240223, |
| "step": 135, |
| "step_time": 25.131237101042643 |
| }, |
| { |
| "clip_ratio/high_max": 0.00150074665240633, |
| "clip_ratio/high_mean": 0.00150074665240633, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00150074665240633, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 831.0, |
| "completions/max_terminated_length": 831.0, |
| "completions/mean_length": 164.31668090820312, |
| "completions/mean_terminated_length": 164.31668090820312, |
| "completions/min_length": 42.0, |
| "completions/min_terminated_length": 42.0, |
| "entropy": 0.3020128731926282, |
| "epoch": 0.3541666666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006879194173961878, |
| "learning_rate": 1e-05, |
| "loss": -0.00875831302255392, |
| "num_tokens": 3198459.0, |
| "reward": 0.5864339470863342, |
| "reward_std": 0.48173201084136963, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.08023274689912796, |
| "rewards/length_penalty/std": 0.05860226973891258, |
| "sampling/importance_sampling_ratio/max": 1.3575568199157715, |
| "sampling/importance_sampling_ratio/mean": 0.9900080561637878, |
| "sampling/importance_sampling_ratio/min": 0.6078801155090332, |
| "sampling/sampling_logp_difference/max": 0.4977775812149048, |
| "sampling/sampling_logp_difference/mean": 0.018817462027072906, |
| "step": 136, |
| "step_time": 9.386668337043375 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010425222426420078, |
| "clip_ratio/high_mean": 0.0010425222426420078, |
| "clip_ratio/low_mean": 0.00012420151324477047, |
| "clip_ratio/low_min": 0.00012420151324477047, |
| "clip_ratio/region_mean": 0.0011667237558867782, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1650.0, |
| "completions/mean_length": 337.5833435058594, |
| "completions/mean_terminated_length": 278.60345458984375, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.3401995698610942, |
| "epoch": 0.3567708333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011913474649190903, |
| "learning_rate": 1e-05, |
| "loss": -0.004462160170078278, |
| "num_tokens": 3223594.0, |
| "reward": 0.4851644039154053, |
| "reward_std": 0.644486665725708, |
| "rewards/correctness/mean": 0.6499999761581421, |
| "rewards/correctness/std": 0.48099473118782043, |
| "rewards/length_penalty/mean": -0.1648356169462204, |
| "rewards/length_penalty/std": 0.24977564811706543, |
| "sampling/importance_sampling_ratio/max": 1.4032944440841675, |
| "sampling/importance_sampling_ratio/mean": 0.987758994102478, |
| "sampling/importance_sampling_ratio/min": 0.675402045249939, |
| "sampling/sampling_logp_difference/max": 0.39244723320007324, |
| "sampling/sampling_logp_difference/mean": 0.020795874297618866, |
| "step": 137, |
| "step_time": 23.452318438095972 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009688327263575047, |
| "clip_ratio/high_mean": 0.0009688327263575047, |
| "clip_ratio/low_mean": 9.657368354965001e-05, |
| "clip_ratio/low_min": 9.657368354965001e-05, |
| "clip_ratio/region_mean": 0.001065406414757793, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1152.0, |
| "completions/mean_length": 319.91668701171875, |
| "completions/mean_terminated_length": 290.6271057128906, |
| "completions/min_length": 98.0, |
| "completions/min_terminated_length": 98.0, |
| "entropy": 0.28134093681971234, |
| "epoch": 0.359375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009714960120618343, |
| "learning_rate": 1e-05, |
| "loss": -0.002933056093752384, |
| "num_tokens": 3248149.0, |
| "reward": 0.3437907099723816, |
| "reward_std": 0.5442456007003784, |
| "rewards/correctness/mean": 0.5, |
| "rewards/correctness/std": 0.5042194724082947, |
| "rewards/length_penalty/mean": -0.1562093049287796, |
| "rewards/length_penalty/std": 0.15698525309562683, |
| "sampling/importance_sampling_ratio/max": 1.45289945602417, |
| "sampling/importance_sampling_ratio/mean": 0.989715576171875, |
| "sampling/importance_sampling_ratio/min": 0.5085936188697815, |
| "sampling/sampling_logp_difference/max": 0.6761059761047363, |
| "sampling/sampling_logp_difference/mean": 0.018162444233894348, |
| "step": 138, |
| "step_time": 23.83872490702197 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005547938684079176, |
| "clip_ratio/high_mean": 0.0005547938684079176, |
| "clip_ratio/low_mean": 0.000403400044888258, |
| "clip_ratio/low_min": 0.000403400044888258, |
| "clip_ratio/region_mean": 0.0009581938987442603, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1776.0, |
| "completions/mean_length": 335.9000244140625, |
| "completions/mean_terminated_length": 306.88134765625, |
| "completions/min_length": 50.0, |
| "completions/min_terminated_length": 50.0, |
| "entropy": 0.28260938078165054, |
| "epoch": 0.3619791666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008651643060147762, |
| "learning_rate": 1e-05, |
| "loss": 0.009136167354881763, |
| "num_tokens": 3274403.0, |
| "reward": 0.43598634004592896, |
| "reward_std": 0.6005921959877014, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.16401366889476776, |
| "rewards/length_penalty/std": 0.18033228814601898, |
| "sampling/importance_sampling_ratio/max": 1.5373297929763794, |
| "sampling/importance_sampling_ratio/mean": 0.9891994595527649, |
| "sampling/importance_sampling_ratio/min": 0.5845726728439331, |
| "sampling/sampling_logp_difference/max": 0.5368741750717163, |
| "sampling/sampling_logp_difference/mean": 0.018368316814303398, |
| "step": 139, |
| "step_time": 24.152534099062905 |
| }, |
| { |
| "clip_ratio/high_max": 0.001558562519979508, |
| "clip_ratio/high_mean": 0.001558562519979508, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.001558562519979508, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1115.0, |
| "completions/max_terminated_length": 1115.0, |
| "completions/mean_length": 290.63336181640625, |
| "completions/mean_terminated_length": 290.63336181640625, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.34687572717666626, |
| "epoch": 0.3645833333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006356269586831331, |
| "learning_rate": 1e-05, |
| "loss": -0.005928201135247946, |
| "num_tokens": 3295291.0, |
| "reward": 0.40808922052383423, |
| "reward_std": 0.5479541420936584, |
| "rewards/correctness/mean": 0.550000011920929, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.14191080629825592, |
| "rewards/length_penalty/std": 0.08714650571346283, |
| "sampling/importance_sampling_ratio/max": 1.4210586547851562, |
| "sampling/importance_sampling_ratio/mean": 0.9878086447715759, |
| "sampling/importance_sampling_ratio/min": 0.6735121011734009, |
| "sampling/sampling_logp_difference/max": 0.3952493667602539, |
| "sampling/sampling_logp_difference/mean": 0.021373523399233818, |
| "step": 140, |
| "step_time": 12.934645050903782 |
| }, |
| { |
| "clip_ratio/high_max": 0.001379395345187125, |
| "clip_ratio/high_mean": 0.001379395345187125, |
| "clip_ratio/low_mean": 0.0001262473282016193, |
| "clip_ratio/low_min": 0.0001262473282016193, |
| "clip_ratio/region_mean": 0.0015056426733887445, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 976.0, |
| "completions/max_terminated_length": 976.0, |
| "completions/mean_length": 208.15000915527344, |
| "completions/mean_terminated_length": 208.15000915527344, |
| "completions/min_length": 65.0, |
| "completions/min_terminated_length": 65.0, |
| "entropy": 0.290947621067365, |
| "epoch": 0.3671875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008567167446017265, |
| "learning_rate": 1e-05, |
| "loss": 0.016681140288710594, |
| "num_tokens": 3312390.0, |
| "reward": 0.5983642935752869, |
| "reward_std": 0.5022408962249756, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.10163573920726776, |
| "rewards/length_penalty/std": 0.08277872949838638, |
| "sampling/importance_sampling_ratio/max": 1.316832423210144, |
| "sampling/importance_sampling_ratio/mean": 0.9892027974128723, |
| "sampling/importance_sampling_ratio/min": 0.6941069960594177, |
| "sampling/sampling_logp_difference/max": 0.36512911319732666, |
| "sampling/sampling_logp_difference/mean": 0.019044499844312668, |
| "step": 141, |
| "step_time": 11.331600727047771 |
| }, |
| { |
| "clip_ratio/high_max": 0.0016000524240856369, |
| "clip_ratio/high_mean": 0.0016000524240856369, |
| "clip_ratio/low_mean": 0.00023087663187955817, |
| "clip_ratio/low_min": 0.00023087663187955817, |
| "clip_ratio/region_mean": 0.0018309290365626414, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1370.0, |
| "completions/max_terminated_length": 1370.0, |
| "completions/mean_length": 420.88336181640625, |
| "completions/mean_terminated_length": 420.88336181640625, |
| "completions/min_length": 121.0, |
| "completions/min_terminated_length": 121.0, |
| "entropy": 0.31516041855017346, |
| "epoch": 0.3697916666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010966254398226738, |
| "learning_rate": 1e-05, |
| "loss": 0.01936233416199684, |
| "num_tokens": 3343713.0, |
| "reward": 0.46115726232528687, |
| "reward_std": 0.5634023547172546, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.20550943911075592, |
| "rewards/length_penalty/std": 0.13497748970985413, |
| "sampling/importance_sampling_ratio/max": 2.1865732669830322, |
| "sampling/importance_sampling_ratio/mean": 0.9888604879379272, |
| "sampling/importance_sampling_ratio/min": 0.6503047943115234, |
| "sampling/sampling_logp_difference/max": 0.782335638999939, |
| "sampling/sampling_logp_difference/mean": 0.019189435988664627, |
| "step": 142, |
| "step_time": 16.50599941215478 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009856185691508774, |
| "clip_ratio/high_mean": 0.0009856185691508774, |
| "clip_ratio/low_mean": 0.00032225797379699844, |
| "clip_ratio/low_min": 0.00032225797379699844, |
| "clip_ratio/region_mean": 0.0013078765477985144, |
| "completions/clipped_ratio": 0.10000000894069672, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1384.0, |
| "completions/mean_length": 377.7500305175781, |
| "completions/mean_terminated_length": 192.1666717529297, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.266576424241066, |
| "epoch": 0.3723958333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00745489401742816, |
| "learning_rate": 1e-05, |
| "loss": 0.001821170561015606, |
| "num_tokens": 3370238.0, |
| "reward": 0.6322184801101685, |
| "reward_std": 0.6626749634742737, |
| "rewards/correctness/mean": 0.8166666626930237, |
| "rewards/correctness/std": 0.39020493626594543, |
| "rewards/length_penalty/mean": -0.1844482421875, |
| "rewards/length_penalty/std": 0.29465141892433167, |
| "sampling/importance_sampling_ratio/max": 1.450861930847168, |
| "sampling/importance_sampling_ratio/mean": 0.9897366166114807, |
| "sampling/importance_sampling_ratio/min": 0.6509119868278503, |
| "sampling/sampling_logp_difference/max": 0.4293808937072754, |
| "sampling/sampling_logp_difference/mean": 0.01765863597393036, |
| "step": 143, |
| "step_time": 23.57354960194789 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011120957060484216, |
| "clip_ratio/high_mean": 0.0011120957060484216, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0011120957060484216, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1608.0, |
| "completions/max_terminated_length": 1608.0, |
| "completions/mean_length": 269.20001220703125, |
| "completions/mean_terminated_length": 269.20001220703125, |
| "completions/min_length": 52.0, |
| "completions/min_terminated_length": 52.0, |
| "entropy": 0.28619813919067383, |
| "epoch": 0.375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00952218845486641, |
| "learning_rate": 1e-05, |
| "loss": 0.02192503958940506, |
| "num_tokens": 3389980.0, |
| "reward": 0.5518880486488342, |
| "reward_std": 0.5071089863777161, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.13144531846046448, |
| "rewards/length_penalty/std": 0.17462965846061707, |
| "sampling/importance_sampling_ratio/max": 1.4095127582550049, |
| "sampling/importance_sampling_ratio/mean": 0.9895171523094177, |
| "sampling/importance_sampling_ratio/min": 0.6638590693473816, |
| "sampling/sampling_logp_difference/max": 0.4096853733062744, |
| "sampling/sampling_logp_difference/mean": 0.01845777966082096, |
| "step": 144, |
| "step_time": 18.157811561133713 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009091018970745305, |
| "clip_ratio/high_mean": 0.0009091018970745305, |
| "clip_ratio/low_mean": 0.00011211981958088775, |
| "clip_ratio/low_min": 0.00011211981958088775, |
| "clip_ratio/region_mean": 0.0010212217166554183, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1335.0, |
| "completions/mean_length": 338.2500305175781, |
| "completions/mean_terminated_length": 309.27117919921875, |
| "completions/min_length": 45.0, |
| "completions/min_terminated_length": 45.0, |
| "entropy": 0.3251095935702324, |
| "epoch": 0.3776041666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011539126746356487, |
| "learning_rate": 1e-05, |
| "loss": -0.018626583740115166, |
| "num_tokens": 3416295.0, |
| "reward": 0.4181722104549408, |
| "reward_std": 0.5938374400138855, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.1651611328125, |
| "rewards/length_penalty/std": 0.17000828683376312, |
| "sampling/importance_sampling_ratio/max": 1.5813007354736328, |
| "sampling/importance_sampling_ratio/mean": 0.9880724549293518, |
| "sampling/importance_sampling_ratio/min": 0.6417348980903625, |
| "sampling/sampling_logp_difference/max": 0.4582477807998657, |
| "sampling/sampling_logp_difference/mean": 0.02063819207251072, |
| "step": 145, |
| "step_time": 24.10411034990102 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010189802439223665, |
| "clip_ratio/high_mean": 0.0010189802439223665, |
| "clip_ratio/low_mean": 0.00013171530493612713, |
| "clip_ratio/low_min": 0.00013171530493612713, |
| "clip_ratio/region_mean": 0.0011506955561344512, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1932.0, |
| "completions/mean_length": 386.16668701171875, |
| "completions/mean_terminated_length": 328.862060546875, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.23338767141103745, |
| "epoch": 0.3802083333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0074458373710513115, |
| "learning_rate": 1e-05, |
| "loss": 0.03246995806694031, |
| "num_tokens": 3443075.0, |
| "reward": 0.19477540254592896, |
| "reward_std": 0.6182466149330139, |
| "rewards/correctness/mean": 0.38333332538604736, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.1885579377412796, |
| "rewards/length_penalty/std": 0.23254674673080444, |
| "sampling/importance_sampling_ratio/max": 1.5565075874328613, |
| "sampling/importance_sampling_ratio/mean": 0.9916983842849731, |
| "sampling/importance_sampling_ratio/min": 0.5947276949882507, |
| "sampling/sampling_logp_difference/max": 0.5196516513824463, |
| "sampling/sampling_logp_difference/mean": 0.015373807400465012, |
| "step": 146, |
| "step_time": 23.528178613865748 |
| }, |
| { |
| "clip_ratio/high_max": 0.002173902622113625, |
| "clip_ratio/high_mean": 0.002173902622113625, |
| "clip_ratio/low_mean": 0.00010016026014151673, |
| "clip_ratio/low_min": 0.00010016026014151673, |
| "clip_ratio/region_mean": 0.002274062872553865, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1882.0, |
| "completions/max_terminated_length": 1882.0, |
| "completions/mean_length": 222.7666778564453, |
| "completions/mean_terminated_length": 222.7666778564453, |
| "completions/min_length": 80.0, |
| "completions/min_terminated_length": 80.0, |
| "entropy": 0.2555236717065175, |
| "epoch": 0.3828125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006344654131680727, |
| "learning_rate": 1e-05, |
| "loss": -0.008334213867783546, |
| "num_tokens": 3460441.0, |
| "reward": 0.7578939199447632, |
| "reward_std": 0.3586113750934601, |
| "rewards/correctness/mean": 0.8666666746139526, |
| "rewards/correctness/std": 0.34280335903167725, |
| "rewards/length_penalty/mean": -0.10877278447151184, |
| "rewards/length_penalty/std": 0.11557597666978836, |
| "sampling/importance_sampling_ratio/max": 1.3970977067947388, |
| "sampling/importance_sampling_ratio/mean": 0.9906747937202454, |
| "sampling/importance_sampling_ratio/min": 0.6410454511642456, |
| "sampling/sampling_logp_difference/max": 0.4446549415588379, |
| "sampling/sampling_logp_difference/mean": 0.01693386398255825, |
| "step": 147, |
| "step_time": 20.515333752613515 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008889571375524005, |
| "clip_ratio/high_mean": 0.0008889571375524005, |
| "clip_ratio/low_mean": 0.00011614401591941714, |
| "clip_ratio/low_min": 0.00011614401591941714, |
| "clip_ratio/region_mean": 0.0010051011534718175, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 939.0, |
| "completions/max_terminated_length": 939.0, |
| "completions/mean_length": 161.15000915527344, |
| "completions/mean_terminated_length": 161.15000915527344, |
| "completions/min_length": 79.0, |
| "completions/min_terminated_length": 79.0, |
| "entropy": 0.24598009636004767, |
| "epoch": 0.3854166666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007655350957065821, |
| "learning_rate": 1e-05, |
| "loss": 0.004562852438539267, |
| "num_tokens": 3473310.0, |
| "reward": 0.6713135242462158, |
| "reward_std": 0.4650338888168335, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.07868652045726776, |
| "rewards/length_penalty/std": 0.056608282029628754, |
| "sampling/importance_sampling_ratio/max": 1.2269755601882935, |
| "sampling/importance_sampling_ratio/mean": 0.9908030033111572, |
| "sampling/importance_sampling_ratio/min": 0.6680235266685486, |
| "sampling/sampling_logp_difference/max": 0.40343189239501953, |
| "sampling/sampling_logp_difference/mean": 0.016687966883182526, |
| "step": 148, |
| "step_time": 10.337043582927436 |
| }, |
| { |
| "clip_ratio/high_max": 0.00044555884232977405, |
| "clip_ratio/high_mean": 0.00044555884232977405, |
| "clip_ratio/low_mean": 0.0001466789617552422, |
| "clip_ratio/low_min": 0.0001466789617552422, |
| "clip_ratio/region_mean": 0.0005922377943837395, |
| "completions/clipped_ratio": 0.13333334028720856, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1379.0, |
| "completions/mean_length": 470.60003662109375, |
| "completions/mean_terminated_length": 227.92308044433594, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.3558978835741679, |
| "epoch": 0.3880208333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006767205893993378, |
| "learning_rate": 1e-05, |
| "loss": 0.0018475591205060482, |
| "num_tokens": 3506066.0, |
| "reward": 0.4535481929779053, |
| "reward_std": 0.7182730436325073, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.22978515923023224, |
| "rewards/length_penalty/std": 0.3183263838291168, |
| "sampling/importance_sampling_ratio/max": 1.7930290699005127, |
| "sampling/importance_sampling_ratio/mean": 0.9878892302513123, |
| "sampling/importance_sampling_ratio/min": 0.3776012659072876, |
| "sampling/sampling_logp_difference/max": 0.9739165306091309, |
| "sampling/sampling_logp_difference/mean": 0.020484600216150284, |
| "step": 149, |
| "step_time": 24.465598567388952 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010748216251765068, |
| "clip_ratio/high_mean": 0.0010748216251765068, |
| "clip_ratio/low_mean": 7.190106650038312e-05, |
| "clip_ratio/low_min": 7.190106650038312e-05, |
| "clip_ratio/region_mean": 0.0011467226868262514, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 831.0, |
| "completions/max_terminated_length": 831.0, |
| "completions/mean_length": 198.9666748046875, |
| "completions/mean_terminated_length": 198.9666748046875, |
| "completions/min_length": 46.0, |
| "completions/min_terminated_length": 46.0, |
| "entropy": 0.24489840616782507, |
| "epoch": 0.390625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006498970091342926, |
| "learning_rate": 1e-05, |
| "loss": 0.010379507206380367, |
| "num_tokens": 3523494.0, |
| "reward": 0.5361816883087158, |
| "reward_std": 0.5206894278526306, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.09715168923139572, |
| "rewards/length_penalty/std": 0.07571697980165482, |
| "sampling/importance_sampling_ratio/max": 1.7602131366729736, |
| "sampling/importance_sampling_ratio/mean": 0.9911451935768127, |
| "sampling/importance_sampling_ratio/min": 0.6789537072181702, |
| "sampling/sampling_logp_difference/max": 0.5654349327087402, |
| "sampling/sampling_logp_difference/mean": 0.016955789178609848, |
| "step": 150, |
| "step_time": 10.470643360167742 |
| }, |
| { |
| "clip_ratio/high_max": 0.00048285970600166667, |
| "clip_ratio/high_mean": 0.00048285970600166667, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00048285970600166667, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 396.0, |
| "completions/max_terminated_length": 396.0, |
| "completions/mean_length": 186.9666748046875, |
| "completions/mean_terminated_length": 186.9666748046875, |
| "completions/min_length": 41.0, |
| "completions/min_terminated_length": 41.0, |
| "entropy": 0.18034610648949942, |
| "epoch": 0.3932291666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006692444905638695, |
| "learning_rate": 1e-05, |
| "loss": -0.005250046029686928, |
| "num_tokens": 3538452.0, |
| "reward": 0.5920410752296448, |
| "reward_std": 0.44770506024360657, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.09129231423139572, |
| "rewards/length_penalty/std": 0.046601347625255585, |
| "sampling/importance_sampling_ratio/max": 1.3859885931015015, |
| "sampling/importance_sampling_ratio/mean": 0.9931260347366333, |
| "sampling/importance_sampling_ratio/min": 0.7180125117301941, |
| "sampling/sampling_logp_difference/max": 0.331268310546875, |
| "sampling/sampling_logp_difference/mean": 0.012593076564371586, |
| "step": 151, |
| "step_time": 5.419737122720107 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008022261560351277, |
| "clip_ratio/high_mean": 0.0008022261560351277, |
| "clip_ratio/low_mean": 0.00012610127062847218, |
| "clip_ratio/low_min": 0.00012610127062847218, |
| "clip_ratio/region_mean": 0.0009283274169623231, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1547.0, |
| "completions/mean_length": 396.38336181640625, |
| "completions/mean_terminated_length": 339.4310302734375, |
| "completions/min_length": 62.0, |
| "completions/min_terminated_length": 62.0, |
| "entropy": 0.30013878643512726, |
| "epoch": 0.3958333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010359244421124458, |
| "learning_rate": 1e-05, |
| "loss": -0.00017593428492546082, |
| "num_tokens": 3567575.0, |
| "reward": 0.3231201469898224, |
| "reward_std": 0.6126262545585632, |
| "rewards/correctness/mean": 0.5166666507720947, |
| "rewards/correctness/std": 0.5039393305778503, |
| "rewards/length_penalty/mean": -0.19354654848575592, |
| "rewards/length_penalty/std": 0.2273566722869873, |
| "sampling/importance_sampling_ratio/max": 1.3856196403503418, |
| "sampling/importance_sampling_ratio/mean": 0.9890786409378052, |
| "sampling/importance_sampling_ratio/min": 0.6644014120101929, |
| "sampling/sampling_logp_difference/max": 0.40886878967285156, |
| "sampling/sampling_logp_difference/mean": 0.019233524799346924, |
| "step": 152, |
| "step_time": 24.7036943314597 |
| }, |
| { |
| "clip_ratio/high_max": 0.001738847548646542, |
| "clip_ratio/high_mean": 0.001738847548646542, |
| "clip_ratio/low_mean": 0.00013520920037990436, |
| "clip_ratio/low_min": 0.00013520920037990436, |
| "clip_ratio/region_mean": 0.0018740567126466583, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1811.0, |
| "completions/mean_length": 661.683349609375, |
| "completions/mean_terminated_length": 588.7192993164062, |
| "completions/min_length": 150.0, |
| "completions/min_terminated_length": 150.0, |
| "entropy": 0.33146199335654575, |
| "epoch": 0.3984375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012678775005042553, |
| "learning_rate": 1e-05, |
| "loss": 0.08487413823604584, |
| "num_tokens": 3611786.0, |
| "reward": 0.2269124537706375, |
| "reward_std": 0.5957501530647278, |
| "rewards/correctness/mean": 0.550000011920929, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.32308757305145264, |
| "rewards/length_penalty/std": 0.26396986842155457, |
| "sampling/importance_sampling_ratio/max": 1.492260217666626, |
| "sampling/importance_sampling_ratio/mean": 0.9885464310646057, |
| "sampling/importance_sampling_ratio/min": 0.6078761219978333, |
| "sampling/sampling_logp_difference/max": 0.4977841377258301, |
| "sampling/sampling_logp_difference/mean": 0.019375475123524666, |
| "step": 153, |
| "step_time": 25.65972848981619 |
| }, |
| { |
| "clip_ratio/high_max": 0.000966934496924902, |
| "clip_ratio/high_mean": 0.000966934496924902, |
| "clip_ratio/low_mean": 0.00012354099696191648, |
| "clip_ratio/low_min": 0.00012354099696191648, |
| "clip_ratio/region_mean": 0.0010904754841855417, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 565.0, |
| "completions/max_terminated_length": 565.0, |
| "completions/mean_length": 258.1500244140625, |
| "completions/mean_terminated_length": 258.1500244140625, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.2445888196428617, |
| "epoch": 0.4010416666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007842191495001316, |
| "learning_rate": 1e-05, |
| "loss": 0.009769807569682598, |
| "num_tokens": 3631365.0, |
| "reward": 0.5072835683822632, |
| "reward_std": 0.5011114478111267, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.12604980170726776, |
| "rewards/length_penalty/std": 0.060861602425575256, |
| "sampling/importance_sampling_ratio/max": 1.5449087619781494, |
| "sampling/importance_sampling_ratio/mean": 0.9914615154266357, |
| "sampling/importance_sampling_ratio/min": 0.6598477363586426, |
| "sampling/sampling_logp_difference/max": 0.4349648952484131, |
| "sampling/sampling_logp_difference/mean": 0.01539693120867014, |
| "step": 154, |
| "step_time": 7.938519381452352 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013767722023961444, |
| "clip_ratio/high_mean": 0.0013767722023961444, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0013767722023961444, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1333.0, |
| "completions/max_terminated_length": 1333.0, |
| "completions/mean_length": 283.4666748046875, |
| "completions/mean_terminated_length": 283.4666748046875, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.3006027589241664, |
| "epoch": 0.4036458333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008753904141485691, |
| "learning_rate": 1e-05, |
| "loss": 0.0038349227979779243, |
| "num_tokens": 3651943.0, |
| "reward": 0.7449219226837158, |
| "reward_std": 0.3456880450248718, |
| "rewards/correctness/mean": 0.8833333253860474, |
| "rewards/correctness/std": 0.32373178005218506, |
| "rewards/length_penalty/mean": -0.13841146230697632, |
| "rewards/length_penalty/std": 0.13020534813404083, |
| "sampling/importance_sampling_ratio/max": 1.3878147602081299, |
| "sampling/importance_sampling_ratio/mean": 0.9892247319221497, |
| "sampling/importance_sampling_ratio/min": 0.6438450813293457, |
| "sampling/sampling_logp_difference/max": 0.44029712677001953, |
| "sampling/sampling_logp_difference/mean": 0.018912719562649727, |
| "step": 155, |
| "step_time": 15.2087720092386 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013064674276392907, |
| "clip_ratio/high_mean": 0.0013064674276392907, |
| "clip_ratio/low_mean": 0.00012532449424422035, |
| "clip_ratio/low_min": 0.00012532449424422035, |
| "clip_ratio/region_mean": 0.0014317919364354263, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1831.0, |
| "completions/max_terminated_length": 1831.0, |
| "completions/mean_length": 399.5666809082031, |
| "completions/mean_terminated_length": 399.5666809082031, |
| "completions/min_length": 77.0, |
| "completions/min_terminated_length": 77.0, |
| "entropy": 0.38203605512777966, |
| "epoch": 0.40625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008092342875897884, |
| "learning_rate": 1e-05, |
| "loss": 0.0012528332881629467, |
| "num_tokens": 3681477.0, |
| "reward": 0.5215657949447632, |
| "reward_std": 0.5965589880943298, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.45441964268684387, |
| "rewards/length_penalty/mean": -0.19510091841220856, |
| "rewards/length_penalty/std": 0.20439039170742035, |
| "sampling/importance_sampling_ratio/max": 1.4126158952713013, |
| "sampling/importance_sampling_ratio/mean": 0.9871541857719421, |
| "sampling/importance_sampling_ratio/min": 0.5785622000694275, |
| "sampling/sampling_logp_difference/max": 0.5472092628479004, |
| "sampling/sampling_logp_difference/mean": 0.022443676367402077, |
| "step": 156, |
| "step_time": 21.863228109898046 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009558483191843455, |
| "clip_ratio/high_mean": 0.0009558483191843455, |
| "clip_ratio/low_mean": 0.0001803797398072978, |
| "clip_ratio/low_min": 0.0001803797398072978, |
| "clip_ratio/region_mean": 0.0011362280541410048, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1621.0, |
| "completions/max_terminated_length": 1621.0, |
| "completions/mean_length": 359.88336181640625, |
| "completions/mean_terminated_length": 359.88336181640625, |
| "completions/min_length": 88.0, |
| "completions/min_terminated_length": 88.0, |
| "entropy": 0.32803257803122204, |
| "epoch": 0.4088541666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008344865404069424, |
| "learning_rate": 1e-05, |
| "loss": 0.0019136876799166203, |
| "num_tokens": 3707270.0, |
| "reward": 0.4242757260799408, |
| "reward_std": 0.5537784099578857, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.17572428286075592, |
| "rewards/length_penalty/std": 0.14189386367797852, |
| "sampling/importance_sampling_ratio/max": 1.3910197019577026, |
| "sampling/importance_sampling_ratio/mean": 0.9885038137435913, |
| "sampling/importance_sampling_ratio/min": 0.10615876317024231, |
| "sampling/sampling_logp_difference/max": 2.2428195476531982, |
| "sampling/sampling_logp_difference/mean": 0.020576095208525658, |
| "step": 157, |
| "step_time": 18.745023934170604 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009742771799210459, |
| "clip_ratio/high_mean": 0.0009742771799210459, |
| "clip_ratio/low_mean": 0.0001522625049498553, |
| "clip_ratio/low_min": 0.0001522625049498553, |
| "clip_ratio/region_mean": 0.0011265396848709013, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 843.0, |
| "completions/max_terminated_length": 843.0, |
| "completions/mean_length": 210.90000915527344, |
| "completions/mean_terminated_length": 210.90000915527344, |
| "completions/min_length": 42.0, |
| "completions/min_terminated_length": 42.0, |
| "entropy": 0.2624708066383998, |
| "epoch": 0.4114583333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0075967269949615, |
| "learning_rate": 1e-05, |
| "loss": 0.004550988785922527, |
| "num_tokens": 3723814.0, |
| "reward": 0.5970215201377869, |
| "reward_std": 0.47245991230010986, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.10297851264476776, |
| "rewards/length_penalty/std": 0.07684464752674103, |
| "sampling/importance_sampling_ratio/max": 1.4769223928451538, |
| "sampling/importance_sampling_ratio/mean": 0.9904353618621826, |
| "sampling/importance_sampling_ratio/min": 0.6593142747879028, |
| "sampling/sampling_logp_difference/max": 0.41655492782592773, |
| "sampling/sampling_logp_difference/mean": 0.016945455223321915, |
| "step": 158, |
| "step_time": 9.911353447008878 |
| }, |
| { |
| "clip_ratio/high_max": 0.001323881617281586, |
| "clip_ratio/high_mean": 0.001323881617281586, |
| "clip_ratio/low_mean": 0.00013066927688972405, |
| "clip_ratio/low_min": 0.00013066927688972405, |
| "clip_ratio/region_mean": 0.0014545508893206716, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1474.0, |
| "completions/max_terminated_length": 1474.0, |
| "completions/mean_length": 290.20001220703125, |
| "completions/mean_terminated_length": 290.20001220703125, |
| "completions/min_length": 79.0, |
| "completions/min_terminated_length": 79.0, |
| "entropy": 0.3215399930874507, |
| "epoch": 0.4140625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007774087134748697, |
| "learning_rate": 1e-05, |
| "loss": -0.001009700819849968, |
| "num_tokens": 3746046.0, |
| "reward": 0.558300793170929, |
| "reward_std": 0.5355238318443298, |
| "rewards/correctness/mean": 0.699999988079071, |
| "rewards/correctness/std": 0.4621247947216034, |
| "rewards/length_penalty/mean": -0.14169922471046448, |
| "rewards/length_penalty/std": 0.13521847128868103, |
| "sampling/importance_sampling_ratio/max": 1.447028398513794, |
| "sampling/importance_sampling_ratio/mean": 0.9887447357177734, |
| "sampling/importance_sampling_ratio/min": 0.6603748202323914, |
| "sampling/sampling_logp_difference/max": 0.4149477481842041, |
| "sampling/sampling_logp_difference/mean": 0.020274106413125992, |
| "step": 159, |
| "step_time": 17.460267372895032 |
| }, |
| { |
| "clip_ratio/high_max": 0.00038247270276769996, |
| "clip_ratio/high_mean": 0.00038247270276769996, |
| "clip_ratio/low_mean": 8.51643659795324e-05, |
| "clip_ratio/low_min": 8.51643659795324e-05, |
| "clip_ratio/region_mean": 0.0004676370687472324, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 807.0, |
| "completions/max_terminated_length": 807.0, |
| "completions/mean_length": 149.7666778564453, |
| "completions/mean_terminated_length": 149.7666778564453, |
| "completions/min_length": 60.0, |
| "completions/min_terminated_length": 60.0, |
| "entropy": 0.26372261842091876, |
| "epoch": 0.4166666666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004888623487204313, |
| "learning_rate": 1e-05, |
| "loss": 0.0008063309360295534, |
| "num_tokens": 3759112.0, |
| "reward": 0.7102051377296448, |
| "reward_std": 0.4586796760559082, |
| "rewards/correctness/mean": 0.7833333611488342, |
| "rewards/correctness/std": 0.41545024514198303, |
| "rewards/length_penalty/mean": -0.07312825322151184, |
| "rewards/length_penalty/std": 0.06102290004491806, |
| "sampling/importance_sampling_ratio/max": 1.4286259412765503, |
| "sampling/importance_sampling_ratio/mean": 0.9906315207481384, |
| "sampling/importance_sampling_ratio/min": 0.6765775680541992, |
| "sampling/sampling_logp_difference/max": 0.39070820808410645, |
| "sampling/sampling_logp_difference/mean": 0.018124476075172424, |
| "step": 160, |
| "step_time": 9.77908792393282 |
| }, |
| { |
| "clip_ratio/high_max": 0.000401165772927925, |
| "clip_ratio/high_mean": 0.000401165772927925, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.000401165772927925, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 360.0, |
| "completions/max_terminated_length": 360.0, |
| "completions/mean_length": 126.50000762939453, |
| "completions/mean_terminated_length": 126.50000762939453, |
| "completions/min_length": 34.0, |
| "completions/min_terminated_length": 34.0, |
| "entropy": 0.18323216835657755, |
| "epoch": 0.4192708333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007231113966554403, |
| "learning_rate": 1e-05, |
| "loss": 0.00022532371804118156, |
| "num_tokens": 3769742.0, |
| "reward": 0.8715658187866211, |
| "reward_std": 0.27111440896987915, |
| "rewards/correctness/mean": 0.9333333373069763, |
| "rewards/correctness/std": 0.2515488862991333, |
| "rewards/length_penalty/mean": -0.061767578125, |
| "rewards/length_penalty/std": 0.04397348687052727, |
| "sampling/importance_sampling_ratio/max": 1.7208596467971802, |
| "sampling/importance_sampling_ratio/mean": 0.9937484860420227, |
| "sampling/importance_sampling_ratio/min": 0.6533620357513428, |
| "sampling/sampling_logp_difference/max": 0.5428240299224854, |
| "sampling/sampling_logp_difference/mean": 0.013013537973165512, |
| "step": 161, |
| "step_time": 4.692062028218061 |
| }, |
| { |
| "clip_ratio/high_max": 0.001488847949076444, |
| "clip_ratio/high_mean": 0.001488847949076444, |
| "clip_ratio/low_mean": 7.821054411275934e-05, |
| "clip_ratio/low_min": 7.821054411275934e-05, |
| "clip_ratio/region_mean": 0.0015670584980398417, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 509.0, |
| "completions/max_terminated_length": 509.0, |
| "completions/mean_length": 196.11668395996094, |
| "completions/mean_terminated_length": 196.11668395996094, |
| "completions/min_length": 112.0, |
| "completions/min_terminated_length": 112.0, |
| "entropy": 0.23024292041858038, |
| "epoch": 0.421875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007893051952123642, |
| "learning_rate": 1e-05, |
| "loss": -0.0037894072011113167, |
| "num_tokens": 3786469.0, |
| "reward": 0.6542399525642395, |
| "reward_std": 0.4389653503894806, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.09576009213924408, |
| "rewards/length_penalty/std": 0.0384553000330925, |
| "sampling/importance_sampling_ratio/max": 1.460600733757019, |
| "sampling/importance_sampling_ratio/mean": 0.9912828803062439, |
| "sampling/importance_sampling_ratio/min": 0.6717548966407776, |
| "sampling/sampling_logp_difference/max": 0.3978617191314697, |
| "sampling/sampling_logp_difference/mean": 0.015774846076965332, |
| "step": 162, |
| "step_time": 6.672657647868618 |
| }, |
| { |
| "clip_ratio/high_max": 0.0004907674350154897, |
| "clip_ratio/high_mean": 0.0004907674350154897, |
| "clip_ratio/low_mean": 0.0002051480890562137, |
| "clip_ratio/low_min": 0.0002051480890562137, |
| "clip_ratio/region_mean": 0.0006959155046691498, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 844.0, |
| "completions/max_terminated_length": 844.0, |
| "completions/mean_length": 173.00001525878906, |
| "completions/mean_terminated_length": 173.00001525878906, |
| "completions/min_length": 53.0, |
| "completions/min_terminated_length": 53.0, |
| "entropy": 0.2069596921404203, |
| "epoch": 0.4244791666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00695077795535326, |
| "learning_rate": 1e-05, |
| "loss": -0.0032284576445817947, |
| "num_tokens": 3800769.0, |
| "reward": 0.8321940302848816, |
| "reward_std": 0.2992183268070221, |
| "rewards/correctness/mean": 0.9166666865348816, |
| "rewards/correctness/std": 0.2787178158760071, |
| "rewards/length_penalty/mean": -0.08447265625, |
| "rewards/length_penalty/std": 0.060042813420295715, |
| "sampling/importance_sampling_ratio/max": 1.4214134216308594, |
| "sampling/importance_sampling_ratio/mean": 0.9929218888282776, |
| "sampling/importance_sampling_ratio/min": 0.6507685780525208, |
| "sampling/sampling_logp_difference/max": 0.42960119247436523, |
| "sampling/sampling_logp_difference/mean": 0.014010408893227577, |
| "step": 163, |
| "step_time": 9.752324955305085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007501483584443728, |
| "clip_ratio/high_mean": 0.0007501483584443728, |
| "clip_ratio/low_mean": 0.0001942134986165911, |
| "clip_ratio/low_min": 0.0001942134986165911, |
| "clip_ratio/region_mean": 0.0009443618667622408, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 876.0, |
| "completions/max_terminated_length": 876.0, |
| "completions/mean_length": 144.4666748046875, |
| "completions/mean_terminated_length": 144.4666748046875, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.22368918607632318, |
| "epoch": 0.4270833333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012669955380260944, |
| "learning_rate": 1e-05, |
| "loss": 0.021695248782634735, |
| "num_tokens": 3812477.0, |
| "reward": 0.8794596791267395, |
| "reward_std": 0.24852481484413147, |
| "rewards/correctness/mean": 0.949999988079071, |
| "rewards/correctness/std": 0.2197841852903366, |
| "rewards/length_penalty/mean": -0.07054036110639572, |
| "rewards/length_penalty/std": 0.05250748246908188, |
| "sampling/importance_sampling_ratio/max": 1.467110276222229, |
| "sampling/importance_sampling_ratio/mean": 0.9919486045837402, |
| "sampling/importance_sampling_ratio/min": 0.5953950881958008, |
| "sampling/sampling_logp_difference/max": 0.5185301303863525, |
| "sampling/sampling_logp_difference/mean": 0.016210168600082397, |
| "step": 164, |
| "step_time": 9.809752298286185 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005854500195709988, |
| "clip_ratio/high_mean": 0.0005854500195709988, |
| "clip_ratio/low_mean": 6.260956676366429e-05, |
| "clip_ratio/low_min": 6.260956676366429e-05, |
| "clip_ratio/region_mean": 0.0006480595863346631, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 659.0, |
| "completions/max_terminated_length": 659.0, |
| "completions/mean_length": 318.4666748046875, |
| "completions/mean_terminated_length": 318.4666748046875, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.24754649152358374, |
| "epoch": 0.4296875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007724328897893429, |
| "learning_rate": 1e-05, |
| "loss": -0.0033492427319288254, |
| "num_tokens": 3836505.0, |
| "reward": 0.2611653804779053, |
| "reward_std": 0.5095491409301758, |
| "rewards/correctness/mean": 0.4166666567325592, |
| "rewards/correctness/std": 0.49716717004776, |
| "rewards/length_penalty/mean": -0.15550130605697632, |
| "rewards/length_penalty/std": 0.07086849957704544, |
| "sampling/importance_sampling_ratio/max": 1.4509283304214478, |
| "sampling/importance_sampling_ratio/mean": 0.9912716746330261, |
| "sampling/importance_sampling_ratio/min": 0.6634112596511841, |
| "sampling/sampling_logp_difference/max": 0.41036009788513184, |
| "sampling/sampling_logp_difference/mean": 0.01627899706363678, |
| "step": 165, |
| "step_time": 8.773001195164397 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002805858578843375, |
| "clip_ratio/high_mean": 0.0002805858578843375, |
| "clip_ratio/low_mean": 0.00019337148599637052, |
| "clip_ratio/low_min": 0.00019337148599637052, |
| "clip_ratio/region_mean": 0.0004739573535819848, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1140.0, |
| "completions/max_terminated_length": 1140.0, |
| "completions/mean_length": 165.6333465576172, |
| "completions/mean_terminated_length": 165.6333465576172, |
| "completions/min_length": 40.0, |
| "completions/min_terminated_length": 40.0, |
| "entropy": 0.25983882943789166, |
| "epoch": 0.4322916666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0067171938717365265, |
| "learning_rate": 1e-05, |
| "loss": 0.006007818505167961, |
| "num_tokens": 3849423.0, |
| "reward": 0.6857910752296448, |
| "reward_std": 0.48303601145744324, |
| "rewards/correctness/mean": 0.7666666507720947, |
| "rewards/correctness/std": 0.42652183771133423, |
| "rewards/length_penalty/mean": -0.08087565004825592, |
| "rewards/length_penalty/std": 0.09182540327310562, |
| "sampling/importance_sampling_ratio/max": 1.4679410457611084, |
| "sampling/importance_sampling_ratio/mean": 0.9908838868141174, |
| "sampling/importance_sampling_ratio/min": 0.66629958152771, |
| "sampling/sampling_logp_difference/max": 0.40601587295532227, |
| "sampling/sampling_logp_difference/mean": 0.017038719728589058, |
| "step": 166, |
| "step_time": 12.892003107815981 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010135714740802844, |
| "clip_ratio/high_mean": 0.0010135714740802844, |
| "clip_ratio/low_mean": 0.00023746319250979772, |
| "clip_ratio/low_min": 0.00023746319250979772, |
| "clip_ratio/region_mean": 0.001251034676291359, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1407.0, |
| "completions/max_terminated_length": 1407.0, |
| "completions/mean_length": 202.10000610351562, |
| "completions/mean_terminated_length": 202.10000610351562, |
| "completions/min_length": 75.0, |
| "completions/min_terminated_length": 75.0, |
| "entropy": 0.3261208087205887, |
| "epoch": 0.4348958333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010386783629655838, |
| "learning_rate": 1e-05, |
| "loss": 0.005768672563135624, |
| "num_tokens": 3865769.0, |
| "reward": 0.5513184070587158, |
| "reward_std": 0.5292786359786987, |
| "rewards/correctness/mean": 0.6499999761581421, |
| "rewards/correctness/std": 0.48099473118782043, |
| "rewards/length_penalty/mean": -0.09868164360523224, |
| "rewards/length_penalty/std": 0.11098555475473404, |
| "sampling/importance_sampling_ratio/max": 1.4042489528656006, |
| "sampling/importance_sampling_ratio/mean": 0.9873855710029602, |
| "sampling/importance_sampling_ratio/min": 0.5515028238296509, |
| "sampling/sampling_logp_difference/max": 0.5951082706451416, |
| "sampling/sampling_logp_difference/mean": 0.022144125774502754, |
| "step": 167, |
| "step_time": 15.778006123844534 |
| }, |
| { |
| "clip_ratio/high_max": 0.0011881113896379247, |
| "clip_ratio/high_mean": 0.0011881113896379247, |
| "clip_ratio/low_mean": 0.00021712371866063526, |
| "clip_ratio/low_min": 0.00021712371866063526, |
| "clip_ratio/region_mean": 0.0014052351034479216, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1961.0, |
| "completions/mean_length": 452.63336181640625, |
| "completions/mean_terminated_length": 368.6666564941406, |
| "completions/min_length": 74.0, |
| "completions/min_terminated_length": 74.0, |
| "entropy": 0.40721839666366577, |
| "epoch": 0.4375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008813354186713696, |
| "learning_rate": 1e-05, |
| "loss": 0.025962810963392258, |
| "num_tokens": 3896947.0, |
| "reward": 0.2289876490831375, |
| "reward_std": 0.6647235155105591, |
| "rewards/correctness/mean": 0.44999998807907104, |
| "rewards/correctness/std": 0.5016920566558838, |
| "rewards/length_penalty/mean": -0.22101236879825592, |
| "rewards/length_penalty/std": 0.2611634433269501, |
| "sampling/importance_sampling_ratio/max": 2.4766576290130615, |
| "sampling/importance_sampling_ratio/mean": 0.9868730902671814, |
| "sampling/importance_sampling_ratio/min": 0.6674465537071228, |
| "sampling/sampling_logp_difference/max": 0.9069099426269531, |
| "sampling/sampling_logp_difference/mean": 0.0224870927631855, |
| "step": 168, |
| "step_time": 23.970061120111495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0018543623543033998, |
| "clip_ratio/high_mean": 0.0018543623543033998, |
| "clip_ratio/low_mean": 0.0002695417885358135, |
| "clip_ratio/low_min": 0.0002695417885358135, |
| "clip_ratio/region_mean": 0.0021239041331379362, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 404.0, |
| "completions/max_terminated_length": 404.0, |
| "completions/mean_length": 207.80001831054688, |
| "completions/mean_terminated_length": 207.80001831054688, |
| "completions/min_length": 78.0, |
| "completions/min_terminated_length": 78.0, |
| "entropy": 0.2653016845385234, |
| "epoch": 0.4401041666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007425842806696892, |
| "learning_rate": 1e-05, |
| "loss": -0.0009765028953552246, |
| "num_tokens": 3913765.0, |
| "reward": 0.7318685054779053, |
| "reward_std": 0.376801997423172, |
| "rewards/correctness/mean": 0.8333333134651184, |
| "rewards/correctness/std": 0.3758230209350586, |
| "rewards/length_penalty/mean": -0.10146484524011612, |
| "rewards/length_penalty/std": 0.04058496654033661, |
| "sampling/importance_sampling_ratio/max": 1.4490848779678345, |
| "sampling/importance_sampling_ratio/mean": 0.9909523725509644, |
| "sampling/importance_sampling_ratio/min": 0.6649330854415894, |
| "sampling/sampling_logp_difference/max": 0.4080688953399658, |
| "sampling/sampling_logp_difference/mean": 0.018061237409710884, |
| "step": 169, |
| "step_time": 6.1799267758615315 |
| }, |
| { |
| "clip_ratio/high_max": 0.0020349578796109804, |
| "clip_ratio/high_mean": 0.0020349578796109804, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0020349578796109804, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1435.0, |
| "completions/max_terminated_length": 1435.0, |
| "completions/mean_length": 321.3500061035156, |
| "completions/mean_terminated_length": 321.3500061035156, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.40403922150532406, |
| "epoch": 0.4427083333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008129854686558247, |
| "learning_rate": 1e-05, |
| "loss": -0.0004943303065374494, |
| "num_tokens": 3935876.0, |
| "reward": 0.42642417550086975, |
| "reward_std": 0.5511415004730225, |
| "rewards/correctness/mean": 0.5833333134651184, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.15690918266773224, |
| "rewards/length_penalty/std": 0.14585080742835999, |
| "sampling/importance_sampling_ratio/max": 1.3994277715682983, |
| "sampling/importance_sampling_ratio/mean": 0.9862880706787109, |
| "sampling/importance_sampling_ratio/min": 0.640641450881958, |
| "sampling/sampling_logp_difference/max": 0.4452853202819824, |
| "sampling/sampling_logp_difference/mean": 0.02334929071366787, |
| "step": 170, |
| "step_time": 16.417588389012963 |
| }, |
| { |
| "clip_ratio/high_max": 0.0005434475412281851, |
| "clip_ratio/high_mean": 0.0005434475412281851, |
| "clip_ratio/low_mean": 0.00021169464647149047, |
| "clip_ratio/low_min": 0.00021169464647149047, |
| "clip_ratio/region_mean": 0.0007551421876996756, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 244.0, |
| "completions/max_terminated_length": 244.0, |
| "completions/mean_length": 152.9166717529297, |
| "completions/mean_terminated_length": 152.9166717529297, |
| "completions/min_length": 88.0, |
| "completions/min_terminated_length": 88.0, |
| "entropy": 0.22362313916285834, |
| "epoch": 0.4453125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.003869452513754368, |
| "learning_rate": 1e-05, |
| "loss": -0.004502504598349333, |
| "num_tokens": 3951121.0, |
| "reward": 0.39200034737586975, |
| "reward_std": 0.497317910194397, |
| "rewards/correctness/mean": 0.46666666865348816, |
| "rewards/correctness/std": 0.5030977725982666, |
| "rewards/length_penalty/mean": -0.0746663436293602, |
| "rewards/length_penalty/std": 0.019079601392149925, |
| "sampling/importance_sampling_ratio/max": 1.42660653591156, |
| "sampling/importance_sampling_ratio/mean": 0.9916558265686035, |
| "sampling/importance_sampling_ratio/min": 0.6406103372573853, |
| "sampling/sampling_logp_difference/max": 0.44533395767211914, |
| "sampling/sampling_logp_difference/mean": 0.015209052711725235, |
| "step": 171, |
| "step_time": 4.34205949306488 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009146964778968444, |
| "clip_ratio/high_mean": 0.0009146964778968444, |
| "clip_ratio/low_mean": 0.00033251283700034645, |
| "clip_ratio/low_min": 0.00033251283700034645, |
| "clip_ratio/region_mean": 0.0012472093124718715, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1502.0, |
| "completions/max_terminated_length": 1502.0, |
| "completions/mean_length": 326.16668701171875, |
| "completions/mean_terminated_length": 326.16668701171875, |
| "completions/min_length": 68.0, |
| "completions/min_terminated_length": 68.0, |
| "entropy": 0.24311373382806778, |
| "epoch": 0.4479166666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008429162204265594, |
| "learning_rate": 1e-05, |
| "loss": -0.0009666156256571412, |
| "num_tokens": 3974661.0, |
| "reward": 0.5240722894668579, |
| "reward_std": 0.5196561813354492, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.1592610627412796, |
| "rewards/length_penalty/std": 0.14245952665805817, |
| "sampling/importance_sampling_ratio/max": 1.3778764009475708, |
| "sampling/importance_sampling_ratio/mean": 0.9916579723358154, |
| "sampling/importance_sampling_ratio/min": 0.7003793120384216, |
| "sampling/sampling_logp_difference/max": 0.35613322257995605, |
| "sampling/sampling_logp_difference/mean": 0.015219391323626041, |
| "step": 172, |
| "step_time": 17.61984484316781 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013613375291849177, |
| "clip_ratio/high_mean": 0.0013613375291849177, |
| "clip_ratio/low_mean": 0.0002175805081302921, |
| "clip_ratio/low_min": 0.0002175805081302921, |
| "clip_ratio/region_mean": 0.0015789180373152096, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 571.0, |
| "completions/max_terminated_length": 571.0, |
| "completions/mean_length": 167.5166778564453, |
| "completions/mean_terminated_length": 167.5166778564453, |
| "completions/min_length": 79.0, |
| "completions/min_terminated_length": 79.0, |
| "entropy": 0.2341864729921023, |
| "epoch": 0.4505208333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005231955554336309, |
| "learning_rate": 1e-05, |
| "loss": -0.002872299635782838, |
| "num_tokens": 3987732.0, |
| "reward": 0.6348714232444763, |
| "reward_std": 0.4755999445915222, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.4544196128845215, |
| "rewards/length_penalty/mean": -0.08179524540901184, |
| "rewards/length_penalty/std": 0.042057525366544724, |
| "sampling/importance_sampling_ratio/max": 1.3777832984924316, |
| "sampling/importance_sampling_ratio/mean": 0.9919395446777344, |
| "sampling/importance_sampling_ratio/min": 0.5426834225654602, |
| "sampling/sampling_logp_difference/max": 0.6112291812896729, |
| "sampling/sampling_logp_difference/mean": 0.015635032206773758, |
| "step": 173, |
| "step_time": 6.961626183940098 |
| }, |
| { |
| "clip_ratio/high_max": 0.000991317894659005, |
| "clip_ratio/high_mean": 0.000991317894659005, |
| "clip_ratio/low_mean": 0.00021105816995259374, |
| "clip_ratio/low_min": 0.00021105816995259374, |
| "clip_ratio/region_mean": 0.0012023760597609605, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1405.0, |
| "completions/max_terminated_length": 1405.0, |
| "completions/mean_length": 302.2500305175781, |
| "completions/mean_terminated_length": 302.2500305175781, |
| "completions/min_length": 83.0, |
| "completions/min_terminated_length": 83.0, |
| "entropy": 0.27552099029223126, |
| "epoch": 0.453125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00842344295233488, |
| "learning_rate": 1e-05, |
| "loss": -0.0068604000844061375, |
| "num_tokens": 4009737.0, |
| "reward": 0.469083696603775, |
| "reward_std": 0.4979739785194397, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.1475830078125, |
| "rewards/length_penalty/std": 0.09452886879444122, |
| "sampling/importance_sampling_ratio/max": 1.4372570514678955, |
| "sampling/importance_sampling_ratio/mean": 0.9905274510383606, |
| "sampling/importance_sampling_ratio/min": 0.6625616550445557, |
| "sampling/sampling_logp_difference/max": 0.41164159774780273, |
| "sampling/sampling_logp_difference/mean": 0.017363766208291054, |
| "step": 174, |
| "step_time": 15.728919385932386 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008217159969111284, |
| "clip_ratio/high_mean": 0.0008217159969111284, |
| "clip_ratio/low_mean": 0.00016037906849912056, |
| "clip_ratio/low_min": 0.00016037906849912056, |
| "clip_ratio/region_mean": 0.0009820950702608873, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1620.0, |
| "completions/mean_length": 423.4833679199219, |
| "completions/mean_terminated_length": 337.9824523925781, |
| "completions/min_length": 139.0, |
| "completions/min_terminated_length": 139.0, |
| "entropy": 0.26755034178495407, |
| "epoch": 0.4557291666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009219520725309849, |
| "learning_rate": 1e-05, |
| "loss": 0.023341964930295944, |
| "num_tokens": 4040156.0, |
| "reward": 0.10988770425319672, |
| "reward_std": 0.5893821120262146, |
| "rewards/correctness/mean": 0.3166666626930237, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.20677897334098816, |
| "rewards/length_penalty/std": 0.2313869148492813, |
| "sampling/importance_sampling_ratio/max": 1.463270664215088, |
| "sampling/importance_sampling_ratio/mean": 0.9901127219200134, |
| "sampling/importance_sampling_ratio/min": 0.5932039618492126, |
| "sampling/sampling_logp_difference/max": 0.5222170352935791, |
| "sampling/sampling_logp_difference/mean": 0.017357870936393738, |
| "step": 175, |
| "step_time": 24.087126932106912 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013210923207225278, |
| "clip_ratio/high_mean": 0.0013210923207225278, |
| "clip_ratio/low_mean": 0.00021111537353135645, |
| "clip_ratio/low_min": 0.00021111537353135645, |
| "clip_ratio/region_mean": 0.0015322076845526074, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 835.0, |
| "completions/max_terminated_length": 835.0, |
| "completions/mean_length": 200.15000915527344, |
| "completions/mean_terminated_length": 200.15000915527344, |
| "completions/min_length": 86.0, |
| "completions/min_terminated_length": 86.0, |
| "entropy": 0.3188545157512029, |
| "epoch": 0.4583333333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007890506647527218, |
| "learning_rate": 1e-05, |
| "loss": 0.014330035075545311, |
| "num_tokens": 4056535.0, |
| "reward": 0.7022705674171448, |
| "reward_std": 0.4308643639087677, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4033755958080292, |
| "rewards/length_penalty/mean": -0.09772948920726776, |
| "rewards/length_penalty/std": 0.07166332006454468, |
| "sampling/importance_sampling_ratio/max": 1.3384337425231934, |
| "sampling/importance_sampling_ratio/mean": 0.9889305233955383, |
| "sampling/importance_sampling_ratio/min": 0.5787335634231567, |
| "sampling/sampling_logp_difference/max": 0.5469131469726562, |
| "sampling/sampling_logp_difference/mean": 0.02052983269095421, |
| "step": 176, |
| "step_time": 9.789259559009224 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010531334749733408, |
| "clip_ratio/high_mean": 0.0010531334749733408, |
| "clip_ratio/low_mean": 6.712310520621638e-05, |
| "clip_ratio/low_min": 6.712310520621638e-05, |
| "clip_ratio/region_mean": 0.0011202565801795572, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 496.0, |
| "completions/max_terminated_length": 496.0, |
| "completions/mean_length": 205.86668395996094, |
| "completions/mean_terminated_length": 205.86668395996094, |
| "completions/min_length": 110.0, |
| "completions/min_terminated_length": 110.0, |
| "entropy": 0.23157701641321182, |
| "epoch": 0.4609375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009520449675619602, |
| "learning_rate": 1e-05, |
| "loss": 0.009514321573078632, |
| "num_tokens": 4072887.0, |
| "reward": 0.682812511920929, |
| "reward_std": 0.45178183913230896, |
| "rewards/correctness/mean": 0.7833333611488342, |
| "rewards/correctness/std": 0.41545021533966064, |
| "rewards/length_penalty/mean": -0.10052083432674408, |
| "rewards/length_penalty/std": 0.0465385727584362, |
| "sampling/importance_sampling_ratio/max": 1.3630462884902954, |
| "sampling/importance_sampling_ratio/mean": 0.9913666844367981, |
| "sampling/importance_sampling_ratio/min": 0.6499996185302734, |
| "sampling/sampling_logp_difference/max": 0.4307835102081299, |
| "sampling/sampling_logp_difference/mean": 0.015578744001686573, |
| "step": 177, |
| "step_time": 7.4569646110758185 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007965431820290784, |
| "clip_ratio/high_mean": 0.0007965431820290784, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0007965431820290784, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1347.0, |
| "completions/mean_length": 369.8000183105469, |
| "completions/mean_terminated_length": 311.9310302734375, |
| "completions/min_length": 146.0, |
| "completions/min_terminated_length": 146.0, |
| "entropy": 0.28369088967641193, |
| "epoch": 0.4635416666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012708203867077827, |
| "learning_rate": 1e-05, |
| "loss": 0.05855375900864601, |
| "num_tokens": 4101665.0, |
| "reward": 0.5861002802848816, |
| "reward_std": 0.5086846351623535, |
| "rewards/correctness/mean": 0.7666666507720947, |
| "rewards/correctness/std": 0.4265218675136566, |
| "rewards/length_penalty/mean": -0.18056640028953552, |
| "rewards/length_penalty/std": 0.18427279591560364, |
| "sampling/importance_sampling_ratio/max": 1.459741234779358, |
| "sampling/importance_sampling_ratio/mean": 0.9904149770736694, |
| "sampling/importance_sampling_ratio/min": 0.6388483643531799, |
| "sampling/sampling_logp_difference/max": 0.4480881690979004, |
| "sampling/sampling_logp_difference/mean": 0.017703067511320114, |
| "step": 178, |
| "step_time": 24.59229546249844 |
| }, |
| { |
| "clip_ratio/high_max": 0.0008133062995815029, |
| "clip_ratio/high_mean": 0.0008133062995815029, |
| "clip_ratio/low_mean": 8.477450076801081e-05, |
| "clip_ratio/low_min": 8.477450076801081e-05, |
| "clip_ratio/region_mean": 0.0008980808003495137, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 373.0, |
| "completions/max_terminated_length": 373.0, |
| "completions/mean_length": 188.58334350585938, |
| "completions/mean_terminated_length": 188.58334350585938, |
| "completions/min_length": 105.0, |
| "completions/min_terminated_length": 105.0, |
| "entropy": 0.2726237326860428, |
| "epoch": 0.4661458333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00755451200529933, |
| "learning_rate": 1e-05, |
| "loss": -0.005787784233689308, |
| "num_tokens": 4117470.0, |
| "reward": 0.6412516832351685, |
| "reward_std": 0.4445783197879791, |
| "rewards/correctness/mean": 0.7333333492279053, |
| "rewards/correctness/std": 0.4459484815597534, |
| "rewards/length_penalty/mean": -0.0920817032456398, |
| "rewards/length_penalty/std": 0.024951918050646782, |
| "sampling/importance_sampling_ratio/max": 1.428837776184082, |
| "sampling/importance_sampling_ratio/mean": 0.9904715418815613, |
| "sampling/importance_sampling_ratio/min": 0.6479266285896301, |
| "sampling/sampling_logp_difference/max": 0.4339778423309326, |
| "sampling/sampling_logp_difference/mean": 0.018075089901685715, |
| "step": 179, |
| "step_time": 5.647872886853293 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009715522658855965, |
| "clip_ratio/high_mean": 0.0009715522658855965, |
| "clip_ratio/low_mean": 0.00019842491747112945, |
| "clip_ratio/low_min": 0.00019842491747112945, |
| "clip_ratio/region_mean": 0.0011699771906326835, |
| "completions/clipped_ratio": 0.0833333358168602, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1759.0, |
| "completions/mean_length": 479.0500183105469, |
| "completions/mean_terminated_length": 336.4181823730469, |
| "completions/min_length": 108.0, |
| "completions/min_terminated_length": 108.0, |
| "entropy": 0.35652025043964386, |
| "epoch": 0.46875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010214319452643394, |
| "learning_rate": 1e-05, |
| "loss": 0.05867590382695198, |
| "num_tokens": 4151063.0, |
| "reward": 0.43275555968284607, |
| "reward_std": 0.7017909288406372, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.23391112685203552, |
| "rewards/length_penalty/std": 0.283832848072052, |
| "sampling/importance_sampling_ratio/max": 1.8386069536209106, |
| "sampling/importance_sampling_ratio/mean": 0.987835705280304, |
| "sampling/importance_sampling_ratio/min": 0.533545732498169, |
| "sampling/sampling_logp_difference/max": 0.6282105445861816, |
| "sampling/sampling_logp_difference/mean": 0.020993832498788834, |
| "step": 180, |
| "step_time": 24.18702670489438 |
| }, |
| { |
| "clip_ratio/high_max": 0.0020316896261647344, |
| "clip_ratio/high_mean": 0.0020316896261647344, |
| "clip_ratio/low_mean": 0.00010861301173766454, |
| "clip_ratio/low_min": 0.00010861301173766454, |
| "clip_ratio/region_mean": 0.002140302637902399, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 887.0, |
| "completions/max_terminated_length": 887.0, |
| "completions/mean_length": 224.2666778564453, |
| "completions/mean_terminated_length": 224.2666778564453, |
| "completions/min_length": 95.0, |
| "completions/min_terminated_length": 95.0, |
| "entropy": 0.3460531532764435, |
| "epoch": 0.4713541666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008629150688648224, |
| "learning_rate": 1e-05, |
| "loss": 0.0070953695103526115, |
| "num_tokens": 4168809.0, |
| "reward": 0.6404948234558105, |
| "reward_std": 0.4495804011821747, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.10950520634651184, |
| "rewards/length_penalty/std": 0.06321557611227036, |
| "sampling/importance_sampling_ratio/max": 1.3401219844818115, |
| "sampling/importance_sampling_ratio/mean": 0.9888230562210083, |
| "sampling/importance_sampling_ratio/min": 0.6294900178909302, |
| "sampling/sampling_logp_difference/max": 0.4628453254699707, |
| "sampling/sampling_logp_difference/mean": 0.021456988528370857, |
| "step": 181, |
| "step_time": 10.387891037622467 |
| }, |
| { |
| "clip_ratio/high_max": 0.0014224624707518767, |
| "clip_ratio/high_mean": 0.0014224624707518767, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0014224624707518767, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 334.0, |
| "completions/max_terminated_length": 334.0, |
| "completions/mean_length": 200.28334045410156, |
| "completions/mean_terminated_length": 200.28334045410156, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.2385458697875341, |
| "epoch": 0.4739583333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00808472279459238, |
| "learning_rate": 1e-05, |
| "loss": 0.0004611872136592865, |
| "num_tokens": 4185016.0, |
| "reward": 0.7022054195404053, |
| "reward_std": 0.41101449728012085, |
| "rewards/correctness/mean": 0.800000011920929, |
| "rewards/correctness/std": 0.4033755958080292, |
| "rewards/length_penalty/mean": -0.09779459983110428, |
| "rewards/length_penalty/std": 0.027668295428156853, |
| "sampling/importance_sampling_ratio/max": 1.395363450050354, |
| "sampling/importance_sampling_ratio/mean": 0.9917981624603271, |
| "sampling/importance_sampling_ratio/min": 0.6605900526046753, |
| "sampling/sampling_logp_difference/max": 0.41462182998657227, |
| "sampling/sampling_logp_difference/mean": 0.0161735936999321, |
| "step": 182, |
| "step_time": 5.356892099836841 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006513536839823549, |
| "clip_ratio/high_mean": 0.0006513536839823549, |
| "clip_ratio/low_mean": 0.00018819608279348662, |
| "clip_ratio/low_min": 0.00018819608279348662, |
| "clip_ratio/region_mean": 0.0008395497667758415, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 432.0, |
| "completions/max_terminated_length": 432.0, |
| "completions/mean_length": 185.15000915527344, |
| "completions/mean_terminated_length": 185.15000915527344, |
| "completions/min_length": 95.0, |
| "completions/min_terminated_length": 95.0, |
| "entropy": 0.23562670747439066, |
| "epoch": 0.4765625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005911108106374741, |
| "learning_rate": 1e-05, |
| "loss": 0.001526092179119587, |
| "num_tokens": 4199605.0, |
| "reward": 0.8095947504043579, |
| "reward_std": 0.3039305508136749, |
| "rewards/correctness/mean": 0.8999999761581421, |
| "rewards/correctness/std": 0.3025316894054413, |
| "rewards/length_penalty/mean": -0.09040527045726776, |
| "rewards/length_penalty/std": 0.036650579422712326, |
| "sampling/importance_sampling_ratio/max": 1.6597732305526733, |
| "sampling/importance_sampling_ratio/mean": 0.9912936091423035, |
| "sampling/importance_sampling_ratio/min": 0.6699170470237732, |
| "sampling/sampling_logp_difference/max": 0.506680965423584, |
| "sampling/sampling_logp_difference/mean": 0.01581893302500248, |
| "step": 183, |
| "step_time": 5.696207208791748 |
| }, |
| { |
| "clip_ratio/high_max": 0.0007011656804631153, |
| "clip_ratio/high_mean": 0.0007011656804631153, |
| "clip_ratio/low_mean": 0.00022340455325320363, |
| "clip_ratio/low_min": 0.00022340455325320363, |
| "clip_ratio/region_mean": 0.000924570233716319, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 515.0, |
| "completions/max_terminated_length": 515.0, |
| "completions/mean_length": 155.8000030517578, |
| "completions/mean_terminated_length": 155.8000030517578, |
| "completions/min_length": 61.0, |
| "completions/min_terminated_length": 61.0, |
| "entropy": 0.2741078684727351, |
| "epoch": 0.4791666666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009305671788752079, |
| "learning_rate": 1e-05, |
| "loss": 0.0041278027929365635, |
| "num_tokens": 4214383.0, |
| "reward": 0.5405924916267395, |
| "reward_std": 0.5033140182495117, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.07607422024011612, |
| "rewards/length_penalty/std": 0.04119933396577835, |
| "sampling/importance_sampling_ratio/max": 1.4254528284072876, |
| "sampling/importance_sampling_ratio/mean": 0.9906982183456421, |
| "sampling/importance_sampling_ratio/min": 0.5515028238296509, |
| "sampling/sampling_logp_difference/max": 0.5951082706451416, |
| "sampling/sampling_logp_difference/mean": 0.018833739683032036, |
| "step": 184, |
| "step_time": 6.7401324857492 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010858871585999925, |
| "clip_ratio/high_mean": 0.0010858871585999925, |
| "clip_ratio/low_mean": 0.0001812879975962763, |
| "clip_ratio/low_min": 0.0001812879975962763, |
| "clip_ratio/region_mean": 0.001267175156196269, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 623.0, |
| "completions/max_terminated_length": 623.0, |
| "completions/mean_length": 197.433349609375, |
| "completions/mean_terminated_length": 197.433349609375, |
| "completions/min_length": 37.0, |
| "completions/min_terminated_length": 37.0, |
| "entropy": 0.3246779590845108, |
| "epoch": 0.4817708333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0058708954602479935, |
| "learning_rate": 1e-05, |
| "loss": -0.003409245051443577, |
| "num_tokens": 4230169.0, |
| "reward": 0.6702637076377869, |
| "reward_std": 0.43587470054626465, |
| "rewards/correctness/mean": 0.7666666507720947, |
| "rewards/correctness/std": 0.42652183771133423, |
| "rewards/length_penalty/mean": -0.09640299528837204, |
| "rewards/length_penalty/std": 0.04905299097299576, |
| "sampling/importance_sampling_ratio/max": 1.6733282804489136, |
| "sampling/importance_sampling_ratio/mean": 0.9884640574455261, |
| "sampling/importance_sampling_ratio/min": 0.6856715679168701, |
| "sampling/sampling_logp_difference/max": 0.5148146152496338, |
| "sampling/sampling_logp_difference/mean": 0.02012818120419979, |
| "step": 185, |
| "step_time": 7.5703852088190615 |
| }, |
| { |
| "clip_ratio/high_max": 0.0012596998712979257, |
| "clip_ratio/high_mean": 0.0012596998712979257, |
| "clip_ratio/low_mean": 0.00021923854850077382, |
| "clip_ratio/low_min": 0.00021923854850077382, |
| "clip_ratio/region_mean": 0.0014789384052467842, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1858.0, |
| "completions/mean_length": 367.0500183105469, |
| "completions/mean_terminated_length": 338.559326171875, |
| "completions/min_length": 36.0, |
| "completions/min_terminated_length": 36.0, |
| "entropy": 0.387447456518809, |
| "epoch": 0.484375, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011788755655288696, |
| "learning_rate": 1e-05, |
| "loss": 0.016219286248087883, |
| "num_tokens": 4256232.0, |
| "reward": 0.5041097402572632, |
| "reward_std": 0.5906460285186768, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.17922362685203552, |
| "rewards/length_penalty/std": 0.23003239929676056, |
| "sampling/importance_sampling_ratio/max": 1.597280502319336, |
| "sampling/importance_sampling_ratio/mean": 0.9872614145278931, |
| "sampling/importance_sampling_ratio/min": 0.4651849865913391, |
| "sampling/sampling_logp_difference/max": 0.7653201818466187, |
| "sampling/sampling_logp_difference/mean": 0.022243699058890343, |
| "step": 186, |
| "step_time": 23.802854794077575 |
| }, |
| { |
| "clip_ratio/high_max": 0.002184967636518801, |
| "clip_ratio/high_mean": 0.002184967636518801, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.002184967636518801, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 336.0, |
| "completions/max_terminated_length": 336.0, |
| "completions/mean_length": 163.60000610351562, |
| "completions/mean_terminated_length": 163.60000610351562, |
| "completions/min_length": 31.0, |
| "completions/min_terminated_length": 31.0, |
| "entropy": 0.29509763916333515, |
| "epoch": 0.4869791666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007903927937150002, |
| "learning_rate": 1e-05, |
| "loss": -0.008337264880537987, |
| "num_tokens": 4269048.0, |
| "reward": 0.4534505307674408, |
| "reward_std": 0.5133340358734131, |
| "rewards/correctness/mean": 0.5333333611488342, |
| "rewards/correctness/std": 0.5030977725982666, |
| "rewards/length_penalty/mean": -0.07988281548023224, |
| "rewards/length_penalty/std": 0.0373755618929863, |
| "sampling/importance_sampling_ratio/max": 1.3931249380111694, |
| "sampling/importance_sampling_ratio/mean": 0.9894994497299194, |
| "sampling/importance_sampling_ratio/min": 0.6619951725006104, |
| "sampling/sampling_logp_difference/max": 0.41249704360961914, |
| "sampling/sampling_logp_difference/mean": 0.01967540569603443, |
| "step": 187, |
| "step_time": 4.668262642109767 |
| }, |
| { |
| "clip_ratio/high_max": 0.0015645795016704749, |
| "clip_ratio/high_mean": 0.0015645795016704749, |
| "clip_ratio/low_mean": 8.582218045679231e-05, |
| "clip_ratio/low_min": 8.582218045679231e-05, |
| "clip_ratio/region_mean": 0.0016504016821272671, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 330.0, |
| "completions/max_terminated_length": 330.0, |
| "completions/mean_length": 158.90000915527344, |
| "completions/mean_terminated_length": 158.90000915527344, |
| "completions/min_length": 51.0, |
| "completions/min_terminated_length": 51.0, |
| "entropy": 0.2328964794675509, |
| "epoch": 0.4895833333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007492752745747566, |
| "learning_rate": 1e-05, |
| "loss": -0.004547375254333019, |
| "num_tokens": 4282592.0, |
| "reward": 0.8390788435935974, |
| "reward_std": 0.281450480222702, |
| "rewards/correctness/mean": 0.9166666865348816, |
| "rewards/correctness/std": 0.2787178158760071, |
| "rewards/length_penalty/mean": -0.07758788764476776, |
| "rewards/length_penalty/std": 0.038592565804719925, |
| "sampling/importance_sampling_ratio/max": 1.6002992391586304, |
| "sampling/importance_sampling_ratio/mean": 0.991621732711792, |
| "sampling/importance_sampling_ratio/min": 0.6554294228553772, |
| "sampling/sampling_logp_difference/max": 0.4701906442642212, |
| "sampling/sampling_logp_difference/mean": 0.016134275123476982, |
| "step": 188, |
| "step_time": 5.178897004807368 |
| }, |
| { |
| "clip_ratio/high_max": 0.0013957567571196705, |
| "clip_ratio/high_mean": 0.0013957567571196705, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0013957567571196705, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1211.0, |
| "completions/max_terminated_length": 1211.0, |
| "completions/mean_length": 174.4166717529297, |
| "completions/mean_terminated_length": 174.4166717529297, |
| "completions/min_length": 98.0, |
| "completions/min_terminated_length": 98.0, |
| "entropy": 0.28425592680772144, |
| "epoch": 0.4921875, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00962707307189703, |
| "learning_rate": 1e-05, |
| "loss": -0.014666496776044369, |
| "num_tokens": 4297217.0, |
| "reward": 0.7815023064613342, |
| "reward_std": 0.3627360165119171, |
| "rewards/correctness/mean": 0.8666666746139526, |
| "rewards/correctness/std": 0.34280332922935486, |
| "rewards/length_penalty/mean": -0.0851643905043602, |
| "rewards/length_penalty/std": 0.07231935858726501, |
| "sampling/importance_sampling_ratio/max": 1.4668527841567993, |
| "sampling/importance_sampling_ratio/mean": 0.9902729988098145, |
| "sampling/importance_sampling_ratio/min": 0.6433727741241455, |
| "sampling/sampling_logp_difference/max": 0.44103097915649414, |
| "sampling/sampling_logp_difference/mean": 0.01984160766005516, |
| "step": 189, |
| "step_time": 13.263055418850854 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010920043035487954, |
| "clip_ratio/high_mean": 0.0010920043035487954, |
| "clip_ratio/low_mean": 3.1152647958757974e-05, |
| "clip_ratio/low_min": 3.1152647958757974e-05, |
| "clip_ratio/region_mean": 0.0011231569466569151, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 713.0, |
| "completions/mean_length": 446.61669921875, |
| "completions/mean_terminated_length": 362.3333435058594, |
| "completions/min_length": 169.0, |
| "completions/min_terminated_length": 169.0, |
| "entropy": 0.3280252069234848, |
| "epoch": 0.4947916666666667, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013572316616773605, |
| "learning_rate": 1e-05, |
| "loss": 0.08345124125480652, |
| "num_tokens": 4328744.0, |
| "reward": 0.39859214425086975, |
| "reward_std": 0.5498757362365723, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014302253723, |
| "rewards/length_penalty/mean": -0.21807454526424408, |
| "rewards/length_penalty/std": 0.19477547705173492, |
| "sampling/importance_sampling_ratio/max": 1.4575377702713013, |
| "sampling/importance_sampling_ratio/mean": 0.9882526993751526, |
| "sampling/importance_sampling_ratio/min": 0.6490767598152161, |
| "sampling/sampling_logp_difference/max": 0.4322042465209961, |
| "sampling/sampling_logp_difference/mean": 0.020178040489554405, |
| "step": 190, |
| "step_time": 24.084194434108213 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006802860686245064, |
| "clip_ratio/high_mean": 0.0006802860686245064, |
| "clip_ratio/low_mean": 0.00016894882719498128, |
| "clip_ratio/low_min": 0.00016894882719498128, |
| "clip_ratio/region_mean": 0.0008492348812675724, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1286.0, |
| "completions/mean_length": 389.2166748046875, |
| "completions/mean_terminated_length": 332.0172424316406, |
| "completions/min_length": 99.0, |
| "completions/min_terminated_length": 99.0, |
| "entropy": 0.31601710617542267, |
| "epoch": 0.4973958333333333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009631474502384663, |
| "learning_rate": 1e-05, |
| "loss": -0.0034431982785463333, |
| "num_tokens": 4355827.0, |
| "reward": 0.44328615069389343, |
| "reward_std": 0.5944100022315979, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.19004720449447632, |
| "rewards/length_penalty/std": 0.20690585672855377, |
| "sampling/importance_sampling_ratio/max": 1.5598375797271729, |
| "sampling/importance_sampling_ratio/mean": 0.9891630411148071, |
| "sampling/importance_sampling_ratio/min": 0.5764176249504089, |
| "sampling/sampling_logp_difference/max": 0.5509228706359863, |
| "sampling/sampling_logp_difference/mean": 0.018540620803833008, |
| "step": 191, |
| "step_time": 23.83205215120688 |
| }, |
| { |
| "clip_ratio/high_max": 0.000835374453648304, |
| "clip_ratio/high_mean": 0.000835374453648304, |
| "clip_ratio/low_mean": 0.00015471211857705688, |
| "clip_ratio/low_min": 0.00015471211857705688, |
| "clip_ratio/region_mean": 0.0009900865698000416, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1823.0, |
| "completions/mean_length": 410.10003662109375, |
| "completions/mean_terminated_length": 382.3389892578125, |
| "completions/min_length": 100.0, |
| "completions/min_terminated_length": 100.0, |
| "entropy": 0.2683347687125206, |
| "epoch": 0.5, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008353208191692829, |
| "learning_rate": 1e-05, |
| "loss": 0.004126677289605141, |
| "num_tokens": 4384503.0, |
| "reward": 0.4330892264842987, |
| "reward_std": 0.6324832439422607, |
| "rewards/correctness/mean": 0.6333333253860474, |
| "rewards/correctness/std": 0.48596110939979553, |
| "rewards/length_penalty/mean": -0.20024414360523224, |
| "rewards/length_penalty/std": 0.2191651463508606, |
| "sampling/importance_sampling_ratio/max": 1.4231841564178467, |
| "sampling/importance_sampling_ratio/mean": 0.9901673793792725, |
| "sampling/importance_sampling_ratio/min": 0.6480928659439087, |
| "sampling/sampling_logp_difference/max": 0.43372130393981934, |
| "sampling/sampling_logp_difference/mean": 0.017023297026753426, |
| "step": 192, |
| "step_time": 23.490768049145117 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009399327294280132, |
| "clip_ratio/high_mean": 0.0009399327294280132, |
| "clip_ratio/low_mean": 0.00038650533436642337, |
| "clip_ratio/low_min": 0.00038650533436642337, |
| "clip_ratio/region_mean": 0.001326438068645075, |
| "completions/clipped_ratio": 0.05000000447034836, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1190.0, |
| "completions/mean_length": 392.2166748046875, |
| "completions/mean_terminated_length": 305.0701904296875, |
| "completions/min_length": 88.0, |
| "completions/min_terminated_length": 88.0, |
| "entropy": 0.442382092277209, |
| "epoch": 0.5026041666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008154982700943947, |
| "learning_rate": 1e-05, |
| "loss": 0.0043773651123046875, |
| "num_tokens": 4413166.0, |
| "reward": 0.22515463829040527, |
| "reward_std": 0.6235041618347168, |
| "rewards/correctness/mean": 0.4166666567325592, |
| "rewards/correctness/std": 0.4971671402454376, |
| "rewards/length_penalty/mean": -0.19151204824447632, |
| "rewards/length_penalty/std": 0.22733677923679352, |
| "sampling/importance_sampling_ratio/max": 1.4115692377090454, |
| "sampling/importance_sampling_ratio/mean": 0.984397292137146, |
| "sampling/importance_sampling_ratio/min": 0.5612049102783203, |
| "sampling/sampling_logp_difference/max": 0.5776691436767578, |
| "sampling/sampling_logp_difference/mean": 0.0263215322047472, |
| "step": 193, |
| "step_time": 24.684919007355347 |
| }, |
| { |
| "clip_ratio/high_max": 0.0009402514745791753, |
| "clip_ratio/high_mean": 0.0009402514745791753, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0009402514745791753, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 342.0, |
| "completions/max_terminated_length": 342.0, |
| "completions/mean_length": 142.1666717529297, |
| "completions/mean_terminated_length": 142.1666717529297, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.24257181088129678, |
| "epoch": 0.5052083333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005891128908842802, |
| "learning_rate": 1e-05, |
| "loss": 0.00279633654281497, |
| "num_tokens": 4425266.0, |
| "reward": 0.5305827260017395, |
| "reward_std": 0.49950745701789856, |
| "rewards/correctness/mean": 0.6000000238418579, |
| "rewards/correctness/std": 0.4940321743488312, |
| "rewards/length_penalty/mean": -0.0694173201918602, |
| "rewards/length_penalty/std": 0.024632425978779793, |
| "sampling/importance_sampling_ratio/max": 1.355822205543518, |
| "sampling/importance_sampling_ratio/mean": 0.9913780689239502, |
| "sampling/importance_sampling_ratio/min": 0.6718780994415283, |
| "sampling/sampling_logp_difference/max": 0.3976783752441406, |
| "sampling/sampling_logp_difference/mean": 0.016589993610978127, |
| "step": 194, |
| "step_time": 4.599929476855323 |
| }, |
| { |
| "clip_ratio/high_max": 0.002177854534238577, |
| "clip_ratio/high_mean": 0.002177854534238577, |
| "clip_ratio/low_mean": 0.00010590092279016972, |
| "clip_ratio/low_min": 0.00010590092279016972, |
| "clip_ratio/region_mean": 0.0022837554570287466, |
| "completions/clipped_ratio": 0.03333333507180214, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1550.0, |
| "completions/mean_length": 353.2833557128906, |
| "completions/mean_terminated_length": 294.8448181152344, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.3703527996937434, |
| "epoch": 0.5078125, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01285611093044281, |
| "learning_rate": 1e-05, |
| "loss": 0.050686415284872055, |
| "num_tokens": 4452843.0, |
| "reward": 0.5441650748252869, |
| "reward_std": 0.5211296081542969, |
| "rewards/correctness/mean": 0.7166666388511658, |
| "rewards/correctness/std": 0.45441964268684387, |
| "rewards/length_penalty/mean": -0.17250162363052368, |
| "rewards/length_penalty/std": 0.197291299700737, |
| "sampling/importance_sampling_ratio/max": 1.4465152025222778, |
| "sampling/importance_sampling_ratio/mean": 0.9866592884063721, |
| "sampling/importance_sampling_ratio/min": 0.656119704246521, |
| "sampling/sampling_logp_difference/max": 0.42141199111938477, |
| "sampling/sampling_logp_difference/mean": 0.02245284616947174, |
| "step": 195, |
| "step_time": 23.790127877146006 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010446557474400227, |
| "clip_ratio/high_mean": 0.0010446557474400227, |
| "clip_ratio/low_mean": 0.00025540447677485645, |
| "clip_ratio/low_min": 0.00025540447677485645, |
| "clip_ratio/region_mean": 0.0013000602096629639, |
| "completions/clipped_ratio": 0.06666667014360428, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 2043.0, |
| "completions/mean_length": 455.9667053222656, |
| "completions/mean_terminated_length": 342.2500305175781, |
| "completions/min_length": 115.0, |
| "completions/min_terminated_length": 115.0, |
| "entropy": 0.33578624327977497, |
| "epoch": 0.5104166666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010537753812968731, |
| "learning_rate": 1e-05, |
| "loss": -0.013097257353365421, |
| "num_tokens": 4484481.0, |
| "reward": 0.34402671456336975, |
| "reward_std": 0.675045907497406, |
| "rewards/correctness/mean": 0.5666666626930237, |
| "rewards/correctness/std": 0.49971744418144226, |
| "rewards/length_penalty/mean": -0.22263997793197632, |
| "rewards/length_penalty/std": 0.30317065119743347, |
| "sampling/importance_sampling_ratio/max": 1.5333874225616455, |
| "sampling/importance_sampling_ratio/mean": 0.988110363483429, |
| "sampling/importance_sampling_ratio/min": 0.4756894111633301, |
| "sampling/sampling_logp_difference/max": 0.7429901361465454, |
| "sampling/sampling_logp_difference/mean": 0.021085353568196297, |
| "step": 196, |
| "step_time": 24.096635987982154 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010522353744211916, |
| "clip_ratio/high_mean": 0.0010522353744211916, |
| "clip_ratio/low_mean": 8.50340099229167e-05, |
| "clip_ratio/low_min": 8.50340099229167e-05, |
| "clip_ratio/region_mean": 0.0011372693746428315, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1700.0, |
| "completions/mean_length": 298.88336181640625, |
| "completions/mean_terminated_length": 269.2372741699219, |
| "completions/min_length": 47.0, |
| "completions/min_terminated_length": 47.0, |
| "entropy": 0.3091815114021301, |
| "epoch": 0.5130208333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009435896761715412, |
| "learning_rate": 1e-05, |
| "loss": 0.05867105722427368, |
| "num_tokens": 4506964.0, |
| "reward": 0.5207275748252869, |
| "reward_std": 0.559025228023529, |
| "rewards/correctness/mean": 0.6666666865348816, |
| "rewards/correctness/std": 0.4753827154636383, |
| "rewards/length_penalty/mean": -0.14593912661075592, |
| "rewards/length_penalty/std": 0.15766774117946625, |
| "sampling/importance_sampling_ratio/max": 1.3723230361938477, |
| "sampling/importance_sampling_ratio/mean": 0.9883788228034973, |
| "sampling/importance_sampling_ratio/min": 0.6758474707603455, |
| "sampling/sampling_logp_difference/max": 0.39178788661956787, |
| "sampling/sampling_logp_difference/mean": 0.02071603573858738, |
| "step": 197, |
| "step_time": 22.96256645862013 |
| }, |
| { |
| "clip_ratio/high_max": 0.0010863260831683874, |
| "clip_ratio/high_mean": 0.0010863260831683874, |
| "clip_ratio/low_mean": 0.00016225702711381018, |
| "clip_ratio/low_min": 0.00016225702711381018, |
| "clip_ratio/region_mean": 0.0012485831102821976, |
| "completions/clipped_ratio": 0.01666666753590107, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1879.0, |
| "completions/mean_length": 377.16668701171875, |
| "completions/mean_terminated_length": 348.84747314453125, |
| "completions/min_length": 110.0, |
| "completions/min_terminated_length": 110.0, |
| "entropy": 0.4080641766389211, |
| "epoch": 0.515625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008708346635103226, |
| "learning_rate": 1e-05, |
| "loss": 0.019376683980226517, |
| "num_tokens": 4534954.0, |
| "reward": 0.4991699457168579, |
| "reward_std": 0.6085314154624939, |
| "rewards/correctness/mean": 0.6833333373069763, |
| "rewards/correctness/std": 0.46910181641578674, |
| "rewards/length_penalty/mean": -0.1841634064912796, |
| "rewards/length_penalty/std": 0.2146603763103485, |
| "sampling/importance_sampling_ratio/max": 1.4300575256347656, |
| "sampling/importance_sampling_ratio/mean": 0.9850609302520752, |
| "sampling/importance_sampling_ratio/min": 0.46999064087867737, |
| "sampling/sampling_logp_difference/max": 0.755042552947998, |
| "sampling/sampling_logp_difference/mean": 0.0247640460729599, |
| "step": 198, |
| "step_time": 24.28067171922885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0019491849622378747, |
| "clip_ratio/high_mean": 0.0019491849622378747, |
| "clip_ratio/low_mean": 0.00014657265758917978, |
| "clip_ratio/low_min": 0.00014657265758917978, |
| "clip_ratio/region_mean": 0.0020957576343789697, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1603.0, |
| "completions/max_terminated_length": 1603.0, |
| "completions/mean_length": 282.9666748046875, |
| "completions/mean_terminated_length": 282.9666748046875, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.3899671733379364, |
| "epoch": 0.5182291666666666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00979206059128046, |
| "learning_rate": 1e-05, |
| "loss": 0.0345764122903347, |
| "num_tokens": 4555942.0, |
| "reward": 0.6118327379226685, |
| "reward_std": 0.4913553297519684, |
| "rewards/correctness/mean": 0.75, |
| "rewards/correctness/std": 0.436666876077652, |
| "rewards/length_penalty/mean": -0.13816732168197632, |
| "rewards/length_penalty/std": 0.1290227174758911, |
| "sampling/importance_sampling_ratio/max": 1.4211622476577759, |
| "sampling/importance_sampling_ratio/mean": 0.985889732837677, |
| "sampling/importance_sampling_ratio/min": 0.3645950257778168, |
| "sampling/sampling_logp_difference/max": 1.0089681148529053, |
| "sampling/sampling_logp_difference/mean": 0.02390696294605732, |
| "step": 199, |
| "step_time": 18.03032245999202 |
| }, |
| { |
| "clip_ratio/high_max": 0.0006166282789005587, |
| "clip_ratio/high_mean": 0.0006166282789005587, |
| "clip_ratio/low_mean": 0.00014296916197054088, |
| "clip_ratio/low_min": 0.00014296916197054088, |
| "clip_ratio/region_mean": 0.0007595974311698228, |
| "completions/clipped_ratio": 0.06666667014360428, |
| "completions/max_length": 2048.0, |
| "completions/max_terminated_length": 1969.0, |
| "completions/mean_length": 481.9000244140625, |
| "completions/mean_terminated_length": 370.0357360839844, |
| "completions/min_length": 78.0, |
| "completions/min_terminated_length": 78.0, |
| "entropy": 0.30196554213762283, |
| "epoch": 0.5208333333333334, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010682770982384682, |
| "learning_rate": 1e-05, |
| "loss": 0.005157994106411934, |
| "num_tokens": 4588386.0, |
| "reward": 0.38136395812034607, |
| "reward_std": 0.6818886399269104, |
| "rewards/correctness/mean": 0.6166666746139526, |
| "rewards/correctness/std": 0.4903014004230499, |
| "rewards/length_penalty/mean": -0.23530273139476776, |
| "rewards/length_penalty/std": 0.3114846646785736, |
| "sampling/importance_sampling_ratio/max": 1.5806714296340942, |
| "sampling/importance_sampling_ratio/mean": 0.9888148903846741, |
| "sampling/importance_sampling_ratio/min": 0.6091993451118469, |
| "sampling/sampling_logp_difference/max": 0.49560976028442383, |
| "sampling/sampling_logp_difference/mean": 0.019149139523506165, |
| "step": 200, |
| "step_time": 24.260671503143385 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 400, |
| "num_input_tokens_seen": 4588386, |
| "num_train_epochs": 2, |
| "save_steps": 50, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 10, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|