| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5543237250554324, |
| "eval_steps": 500, |
| "global_step": 2000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2441.0, |
| "completions/max_terminated_length": 1929.0, |
| "completions/mean_length": 475.20078125, |
| "completions/mean_terminated_length": 465.0972076416016, |
| "completions/min_length": 33.7, |
| "completions/min_terminated_length": 33.7, |
| "entropy": 0.28767133336514233, |
| "epoch": 0.005543237250554324, |
| "frac_reward_zero_std": 0.2, |
| "grad_norm": 1.5859375, |
| "learning_rate": 9.9991e-06, |
| "loss": -0.0041, |
| "num_tokens": 910073.0, |
| "reward": 0.3241015747189522, |
| "reward_std": 0.4225256651639938, |
| "rewards/reward_accuracy/mean": 0.2359375, |
| "rewards/reward_accuracy/std": 0.41777620613574984, |
| "rewards/reward_format/mean": 0.0881640650331974, |
| "rewards/reward_format/std": 0.025013190880417823, |
| "sampling/importance_sampling_ratio/max": 2.7370090246200562, |
| "sampling/importance_sampling_ratio/mean": 0.691376370191574, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7040550708770752, |
| "sampling/sampling_logp_difference/mean": 0.015001642610877752, |
| "step": 10, |
| "step_time": 22.58667814009823 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2533.3, |
| "completions/max_terminated_length": 1830.4, |
| "completions/mean_length": 473.46015625, |
| "completions/mean_terminated_length": 465.2191528320312, |
| "completions/min_length": 58.0, |
| "completions/min_terminated_length": 58.0, |
| "entropy": 0.2808842334896326, |
| "epoch": 0.011086474501108648, |
| "frac_reward_zero_std": 0.225, |
| "grad_norm": 0.390625, |
| "learning_rate": 9.9981e-06, |
| "loss": -0.0273, |
| "num_tokens": 1815862.0, |
| "reward": 0.3047265648841858, |
| "reward_std": 0.407260873913765, |
| "rewards/reward_accuracy/mean": 0.21484375, |
| "rewards/reward_accuracy/std": 0.4038012385368347, |
| "rewards/reward_format/mean": 0.08988281413912773, |
| "rewards/reward_format/std": 0.023710903152823447, |
| "sampling/importance_sampling_ratio/max": 2.7636337757110594, |
| "sampling/importance_sampling_ratio/mean": 0.6966065168380737, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7200770735740661, |
| "sampling/sampling_logp_difference/mean": 0.015099686849862338, |
| "step": 20, |
| "step_time": 23.359830860281363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 2148.7, |
| "completions/max_terminated_length": 1913.0, |
| "completions/mean_length": 484.9875, |
| "completions/mean_terminated_length": 478.9272003173828, |
| "completions/min_length": 43.0, |
| "completions/min_terminated_length": 43.0, |
| "entropy": 0.2783487424254417, |
| "epoch": 0.01662971175166297, |
| "frac_reward_zero_std": 0.2875, |
| "grad_norm": 0.4296875, |
| "learning_rate": 9.997100000000001e-06, |
| "loss": -0.038, |
| "num_tokens": 2736398.0, |
| "reward": 0.2952343851327896, |
| "reward_std": 0.39903710782527924, |
| "rewards/reward_accuracy/mean": 0.20390625, |
| "rewards/reward_accuracy/std": 0.3961195766925812, |
| "rewards/reward_format/mean": 0.09132812544703484, |
| "rewards/reward_format/std": 0.021452920511364937, |
| "sampling/importance_sampling_ratio/max": 2.791772389411926, |
| "sampling/importance_sampling_ratio/mean": 0.6881749212741852, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7920448064804078, |
| "sampling/sampling_logp_difference/mean": 0.014975472819060087, |
| "step": 30, |
| "step_time": 20.001702073868366 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2699.7, |
| "completions/max_terminated_length": 1715.6, |
| "completions/mean_length": 457.17890625, |
| "completions/mean_terminated_length": 442.80861206054686, |
| "completions/min_length": 54.9, |
| "completions/min_terminated_length": 54.9, |
| "entropy": 0.25608821865171194, |
| "epoch": 0.022172949002217297, |
| "frac_reward_zero_std": 0.31875, |
| "grad_norm": 0.392578125, |
| "learning_rate": 9.9961e-06, |
| "loss": -0.0197, |
| "num_tokens": 3624075.0, |
| "reward": 0.3400781363248825, |
| "reward_std": 0.42409199476242065, |
| "rewards/reward_accuracy/mean": 0.24765625, |
| "rewards/reward_accuracy/std": 0.4204283744096756, |
| "rewards/reward_format/mean": 0.09242187589406967, |
| "rewards/reward_format/std": 0.021873819082975386, |
| "sampling/importance_sampling_ratio/max": 2.743203854560852, |
| "sampling/importance_sampling_ratio/mean": 0.7255253911018371, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6652591586112976, |
| "sampling/sampling_logp_difference/mean": 0.013892047852277756, |
| "step": 40, |
| "step_time": 25.067468155408278 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2664.7, |
| "completions/max_terminated_length": 1684.5, |
| "completions/mean_length": 525.84296875, |
| "completions/mean_terminated_length": 507.83873291015624, |
| "completions/min_length": 63.7, |
| "completions/min_terminated_length": 63.7, |
| "entropy": 0.2690406873822212, |
| "epoch": 0.02771618625277162, |
| "frac_reward_zero_std": 0.325, |
| "grad_norm": 0.2421875, |
| "learning_rate": 9.9951e-06, |
| "loss": -0.0123, |
| "num_tokens": 4603970.0, |
| "reward": 0.30214844942092894, |
| "reward_std": 0.40420193672180177, |
| "rewards/reward_accuracy/mean": 0.21015625, |
| "rewards/reward_accuracy/std": 0.4002250850200653, |
| "rewards/reward_format/mean": 0.0919921875, |
| "rewards/reward_format/std": 0.022420401498675347, |
| "sampling/importance_sampling_ratio/max": 2.852541518211365, |
| "sampling/importance_sampling_ratio/mean": 0.6980676710605621, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6754738926887512, |
| "sampling/sampling_logp_difference/mean": 0.01431960817426443, |
| "step": 50, |
| "step_time": 25.056258158339187 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2341.7, |
| "completions/max_terminated_length": 1944.8, |
| "completions/mean_length": 467.95625, |
| "completions/mean_terminated_length": 459.8088439941406, |
| "completions/min_length": 54.0, |
| "completions/min_terminated_length": 54.0, |
| "entropy": 0.26547709610313175, |
| "epoch": 0.03325942350332594, |
| "frac_reward_zero_std": 0.29375, |
| "grad_norm": 0.412109375, |
| "learning_rate": 9.994100000000001e-06, |
| "loss": -0.0122, |
| "num_tokens": 5511162.0, |
| "reward": 0.3064062625169754, |
| "reward_std": 0.4033316642045975, |
| "rewards/reward_accuracy/mean": 0.2140625, |
| "rewards/reward_accuracy/std": 0.40034096837043764, |
| "rewards/reward_format/mean": 0.09234375134110451, |
| "rewards/reward_format/std": 0.02109913844615221, |
| "sampling/importance_sampling_ratio/max": 2.8833245277404784, |
| "sampling/importance_sampling_ratio/mean": 0.7393885910511017, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7540618777275085, |
| "sampling/sampling_logp_difference/mean": 0.014592802245169878, |
| "step": 60, |
| "step_time": 21.817739351838828 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2277.7, |
| "completions/max_terminated_length": 1660.2, |
| "completions/mean_length": 484.21328125, |
| "completions/mean_terminated_length": 474.0781280517578, |
| "completions/min_length": 71.5, |
| "completions/min_terminated_length": 71.5, |
| "entropy": 0.25473210504278543, |
| "epoch": 0.038802660753880266, |
| "frac_reward_zero_std": 0.36875, |
| "grad_norm": 0.265625, |
| "learning_rate": 9.993100000000001e-06, |
| "loss": -0.0096, |
| "num_tokens": 6434507.0, |
| "reward": 0.3319921940565109, |
| "reward_std": 0.4242938309907913, |
| "rewards/reward_accuracy/mean": 0.2390625, |
| "rewards/reward_accuracy/std": 0.4218552470207214, |
| "rewards/reward_format/mean": 0.0929296888411045, |
| "rewards/reward_format/std": 0.018039879389107227, |
| "sampling/importance_sampling_ratio/max": 2.797001099586487, |
| "sampling/importance_sampling_ratio/mean": 0.7427074790000916, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6378371357917786, |
| "sampling/sampling_logp_difference/mean": 0.013627664931118489, |
| "step": 70, |
| "step_time": 21.07613220331259 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2226.9, |
| "completions/max_terminated_length": 1744.2, |
| "completions/mean_length": 495.84921875, |
| "completions/mean_terminated_length": 483.62421875, |
| "completions/min_length": 63.7, |
| "completions/min_terminated_length": 63.7, |
| "entropy": 0.25530271921306846, |
| "epoch": 0.04434589800443459, |
| "frac_reward_zero_std": 0.4, |
| "grad_norm": 0.4765625, |
| "learning_rate": 9.9921e-06, |
| "loss": -0.0092, |
| "num_tokens": 7376986.0, |
| "reward": 0.34902345240116117, |
| "reward_std": 0.438951113820076, |
| "rewards/reward_accuracy/mean": 0.2578125, |
| "rewards/reward_accuracy/std": 0.43568557500839233, |
| "rewards/reward_format/mean": 0.09121094048023223, |
| "rewards/reward_format/std": 0.022107023373246194, |
| "sampling/importance_sampling_ratio/max": 2.776185154914856, |
| "sampling/importance_sampling_ratio/mean": 0.7283871471881866, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7127304673194885, |
| "sampling/sampling_logp_difference/mean": 0.013848797604441642, |
| "step": 80, |
| "step_time": 21.350466054305436 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2303.6, |
| "completions/max_terminated_length": 1811.9, |
| "completions/mean_length": 483.6859375, |
| "completions/mean_terminated_length": 475.54478149414064, |
| "completions/min_length": 64.0, |
| "completions/min_terminated_length": 64.0, |
| "entropy": 0.2562675965949893, |
| "epoch": 0.04988913525498891, |
| "frac_reward_zero_std": 0.3625, |
| "grad_norm": 0.27734375, |
| "learning_rate": 9.991100000000002e-06, |
| "loss": -0.0131, |
| "num_tokens": 8291680.0, |
| "reward": 0.3769531399011612, |
| "reward_std": 0.4476018935441971, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.44451017677783966, |
| "rewards/reward_format/mean": 0.094140625, |
| "rewards/reward_format/std": 0.017788084875792264, |
| "sampling/importance_sampling_ratio/max": 2.632981538772583, |
| "sampling/importance_sampling_ratio/mean": 0.7124887049198151, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6937095522880554, |
| "sampling/sampling_logp_difference/mean": 0.0139068104326725, |
| "step": 90, |
| "step_time": 21.44362189238891 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 2356.4, |
| "completions/max_terminated_length": 1910.4, |
| "completions/mean_length": 489.6234375, |
| "completions/mean_terminated_length": 483.5686309814453, |
| "completions/min_length": 87.0, |
| "completions/min_terminated_length": 87.0, |
| "entropy": 0.24633241342380643, |
| "epoch": 0.05543237250554324, |
| "frac_reward_zero_std": 0.4125, |
| "grad_norm": 0.37890625, |
| "learning_rate": 9.990100000000001e-06, |
| "loss": -0.0105, |
| "num_tokens": 9213430.0, |
| "reward": 0.3665234446525574, |
| "reward_std": 0.44464872777462006, |
| "rewards/reward_accuracy/mean": 0.27265625, |
| "rewards/reward_accuracy/std": 0.4416452795267105, |
| "rewards/reward_format/mean": 0.09386719092726707, |
| "rewards/reward_format/std": 0.01875109039247036, |
| "sampling/importance_sampling_ratio/max": 2.8734532833099364, |
| "sampling/importance_sampling_ratio/mean": 0.746452808380127, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7752881407737732, |
| "sampling/sampling_logp_difference/mean": 0.013408410269767046, |
| "step": 100, |
| "step_time": 21.898679677629843 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2570.2, |
| "completions/max_terminated_length": 2060.7, |
| "completions/mean_length": 473.134375, |
| "completions/mean_terminated_length": 461.02857971191406, |
| "completions/min_length": 68.7, |
| "completions/min_terminated_length": 68.7, |
| "entropy": 0.24929247805848717, |
| "epoch": 0.06097560975609756, |
| "frac_reward_zero_std": 0.45625, |
| "grad_norm": 0.1728515625, |
| "learning_rate": 9.9891e-06, |
| "loss": -0.0167, |
| "num_tokens": 10115330.0, |
| "reward": 0.332929702103138, |
| "reward_std": 0.41743423640727995, |
| "rewards/reward_accuracy/mean": 0.2390625, |
| "rewards/reward_accuracy/std": 0.415049684047699, |
| "rewards/reward_format/mean": 0.09386718794703483, |
| "rewards/reward_format/std": 0.0180061474442482, |
| "sampling/importance_sampling_ratio/max": 2.815832567214966, |
| "sampling/importance_sampling_ratio/mean": 0.7657078742980957, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.8004599332809448, |
| "sampling/sampling_logp_difference/mean": 0.013448974210768938, |
| "step": 110, |
| "step_time": 24.150640684505923 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 2309.1, |
| "completions/max_terminated_length": 2090.6, |
| "completions/mean_length": 493.2703125, |
| "completions/mean_terminated_length": 489.13050537109376, |
| "completions/min_length": 80.1, |
| "completions/min_terminated_length": 80.1, |
| "entropy": 0.25036389799788594, |
| "epoch": 0.06651884700665188, |
| "frac_reward_zero_std": 0.40625, |
| "grad_norm": 0.1953125, |
| "learning_rate": 9.9881e-06, |
| "loss": -0.0147, |
| "num_tokens": 11052028.0, |
| "reward": 0.33492189049720766, |
| "reward_std": 0.4206922948360443, |
| "rewards/reward_accuracy/mean": 0.24140625, |
| "rewards/reward_accuracy/std": 0.4188133865594864, |
| "rewards/reward_format/mean": 0.09351562932133675, |
| "rewards/reward_format/std": 0.01887950785458088, |
| "sampling/importance_sampling_ratio/max": 2.7600017547607423, |
| "sampling/importance_sampling_ratio/mean": 0.7368011176586151, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6842268288135529, |
| "sampling/sampling_logp_difference/mean": 0.013709675706923007, |
| "step": 120, |
| "step_time": 21.65553494482301 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2273.4, |
| "completions/max_terminated_length": 1872.1, |
| "completions/mean_length": 537.9296875, |
| "completions/mean_terminated_length": 528.0518981933594, |
| "completions/min_length": 92.9, |
| "completions/min_terminated_length": 92.9, |
| "entropy": 0.23474841229617596, |
| "epoch": 0.07206208425720621, |
| "frac_reward_zero_std": 0.41875, |
| "grad_norm": 0.349609375, |
| "learning_rate": 9.987100000000001e-06, |
| "loss": -0.015, |
| "num_tokens": 12046666.0, |
| "reward": 0.36226564049720766, |
| "reward_std": 0.44047539234161376, |
| "rewards/reward_accuracy/mean": 0.26796875, |
| "rewards/reward_accuracy/std": 0.43751579225063325, |
| "rewards/reward_format/mean": 0.09429687708616256, |
| "rewards/reward_format/std": 0.017854578606784344, |
| "sampling/importance_sampling_ratio/max": 2.7526787519454956, |
| "sampling/importance_sampling_ratio/mean": 0.7333716332912446, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.663422840833664, |
| "sampling/sampling_logp_difference/mean": 0.012724371068179608, |
| "step": 130, |
| "step_time": 21.731316118407996 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2263.5, |
| "completions/max_terminated_length": 2021.5, |
| "completions/mean_length": 504.6109375, |
| "completions/mean_terminated_length": 496.6211791992188, |
| "completions/min_length": 88.4, |
| "completions/min_terminated_length": 88.4, |
| "entropy": 0.2535966087132692, |
| "epoch": 0.07760532150776053, |
| "frac_reward_zero_std": 0.4125, |
| "grad_norm": 0.14453125, |
| "learning_rate": 9.9861e-06, |
| "loss": -0.0055, |
| "num_tokens": 12992728.0, |
| "reward": 0.32839844971895216, |
| "reward_std": 0.4146625339984894, |
| "rewards/reward_accuracy/mean": 0.23359375, |
| "rewards/reward_accuracy/std": 0.4122250974178314, |
| "rewards/reward_format/mean": 0.09480468779802323, |
| "rewards/reward_format/std": 0.01744569381698966, |
| "sampling/importance_sampling_ratio/max": 2.7898416042327883, |
| "sampling/importance_sampling_ratio/mean": 0.7135927855968476, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.283564668893814, |
| "sampling/sampling_logp_difference/mean": 0.01384265935048461, |
| "step": 140, |
| "step_time": 21.59592035859823 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 2163.2, |
| "completions/max_terminated_length": 1949.5, |
| "completions/mean_length": 506.22109375, |
| "completions/mean_terminated_length": 500.3979797363281, |
| "completions/min_length": 66.0, |
| "completions/min_terminated_length": 66.0, |
| "entropy": 0.24640185683965682, |
| "epoch": 0.08314855875831485, |
| "frac_reward_zero_std": 0.41875, |
| "grad_norm": 0.1748046875, |
| "learning_rate": 9.9851e-06, |
| "loss": -0.0201, |
| "num_tokens": 13945651.0, |
| "reward": 0.35082031935453417, |
| "reward_std": 0.43044011294841766, |
| "rewards/reward_accuracy/mean": 0.25703125, |
| "rewards/reward_accuracy/std": 0.42754030227661133, |
| "rewards/reward_format/mean": 0.09378906562924386, |
| "rewards/reward_format/std": 0.019111134577542543, |
| "sampling/importance_sampling_ratio/max": 2.72128746509552, |
| "sampling/importance_sampling_ratio/mean": 0.7165862381458282, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6761679470539093, |
| "sampling/sampling_logp_difference/mean": 0.013467768020927907, |
| "step": 150, |
| "step_time": 20.457572038611396 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2370.3, |
| "completions/max_terminated_length": 2021.6, |
| "completions/mean_length": 510.1578125, |
| "completions/mean_terminated_length": 502.1669067382812, |
| "completions/min_length": 70.4, |
| "completions/min_terminated_length": 70.4, |
| "entropy": 0.23929457310587168, |
| "epoch": 0.08869179600886919, |
| "frac_reward_zero_std": 0.4125, |
| "grad_norm": 0.17578125, |
| "learning_rate": 9.984100000000002e-06, |
| "loss": 0.0124, |
| "num_tokens": 14904453.0, |
| "reward": 0.32101563811302186, |
| "reward_std": 0.41340243220329287, |
| "rewards/reward_accuracy/mean": 0.22734375, |
| "rewards/reward_accuracy/std": 0.41052471101284027, |
| "rewards/reward_format/mean": 0.09367187693715096, |
| "rewards/reward_format/std": 0.019695294555276632, |
| "sampling/importance_sampling_ratio/max": 2.77149817943573, |
| "sampling/importance_sampling_ratio/mean": 0.7340317726135254, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6312026619911194, |
| "sampling/sampling_logp_difference/mean": 0.01297246441245079, |
| "step": 160, |
| "step_time": 21.894550693035125 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2584.4, |
| "completions/max_terminated_length": 2204.3, |
| "completions/mean_length": 493.9609375, |
| "completions/mean_terminated_length": 485.8755218505859, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.22358188461512327, |
| "epoch": 0.0942350332594235, |
| "frac_reward_zero_std": 0.38125, |
| "grad_norm": 0.240234375, |
| "learning_rate": 9.983100000000001e-06, |
| "loss": 0.0078, |
| "num_tokens": 15829307.0, |
| "reward": 0.4012890726327896, |
| "reward_std": 0.4497102588415146, |
| "rewards/reward_accuracy/mean": 0.3078125, |
| "rewards/reward_accuracy/std": 0.44640993475914004, |
| "rewards/reward_format/mean": 0.09347656220197678, |
| "rewards/reward_format/std": 0.018597512412816285, |
| "sampling/importance_sampling_ratio/max": 2.818800449371338, |
| "sampling/importance_sampling_ratio/mean": 0.7718528687953949, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.636805635690689, |
| "sampling/sampling_logp_difference/mean": 0.012335697934031487, |
| "step": 170, |
| "step_time": 23.842233613785357 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2185.1, |
| "completions/max_terminated_length": 1771.7, |
| "completions/mean_length": 497.11015625, |
| "completions/mean_terminated_length": 488.96946105957034, |
| "completions/min_length": 69.9, |
| "completions/min_terminated_length": 69.9, |
| "entropy": 0.23560391748324036, |
| "epoch": 0.09977827050997783, |
| "frac_reward_zero_std": 0.39375, |
| "grad_norm": 0.2060546875, |
| "learning_rate": 9.9821e-06, |
| "loss": -0.0107, |
| "num_tokens": 16777416.0, |
| "reward": 0.31519532203674316, |
| "reward_std": 0.4047440826892853, |
| "rewards/reward_accuracy/mean": 0.22109375, |
| "rewards/reward_accuracy/std": 0.4030202478170395, |
| "rewards/reward_format/mean": 0.09410156533122063, |
| "rewards/reward_format/std": 0.01854334268718958, |
| "sampling/importance_sampling_ratio/max": 2.789047622680664, |
| "sampling/importance_sampling_ratio/mean": 0.7319223642349243, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7251484513282775, |
| "sampling/sampling_logp_difference/mean": 0.012823521625250578, |
| "step": 180, |
| "step_time": 20.84519771062769 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2604.9, |
| "completions/max_terminated_length": 2034.1, |
| "completions/mean_length": 543.1359375, |
| "completions/mean_terminated_length": 523.2943572998047, |
| "completions/min_length": 70.1, |
| "completions/min_terminated_length": 70.1, |
| "entropy": 0.23324271077290176, |
| "epoch": 0.10532150776053215, |
| "frac_reward_zero_std": 0.4, |
| "grad_norm": 0.1884765625, |
| "learning_rate": 9.981100000000002e-06, |
| "loss": -0.0243, |
| "num_tokens": 17774670.0, |
| "reward": 0.35363282561302184, |
| "reward_std": 0.43921445310115814, |
| "rewards/reward_accuracy/mean": 0.26015625, |
| "rewards/reward_accuracy/std": 0.43574453592300416, |
| "rewards/reward_format/mean": 0.09347656294703484, |
| "rewards/reward_format/std": 0.019688358809798957, |
| "sampling/importance_sampling_ratio/max": 2.832689118385315, |
| "sampling/importance_sampling_ratio/mean": 0.7418004214763642, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.083196997642517, |
| "sampling/sampling_logp_difference/mean": 0.012459208071231843, |
| "step": 190, |
| "step_time": 24.192911703372374 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2150.3, |
| "completions/max_terminated_length": 1851.9, |
| "completions/mean_length": 527.4234375, |
| "completions/mean_terminated_length": 519.5043334960938, |
| "completions/min_length": 72.9, |
| "completions/min_terminated_length": 72.9, |
| "entropy": 0.23670608242973684, |
| "epoch": 0.11086474501108648, |
| "frac_reward_zero_std": 0.44375, |
| "grad_norm": 0.2255859375, |
| "learning_rate": 9.9801e-06, |
| "loss": -0.0184, |
| "num_tokens": 18746204.0, |
| "reward": 0.3475781351327896, |
| "reward_std": 0.43367789685726166, |
| "rewards/reward_accuracy/mean": 0.25234375, |
| "rewards/reward_accuracy/std": 0.4323807328939438, |
| "rewards/reward_format/mean": 0.09523437619209289, |
| "rewards/reward_format/std": 0.015932231862097978, |
| "sampling/importance_sampling_ratio/max": 2.813192367553711, |
| "sampling/importance_sampling_ratio/mean": 0.7615881383419036, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.725534725189209, |
| "sampling/sampling_logp_difference/mean": 0.012926409021019936, |
| "step": 200, |
| "step_time": 20.480604119505735 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2153.0, |
| "completions/max_terminated_length": 2030.1, |
| "completions/mean_length": 503.02578125, |
| "completions/mean_terminated_length": 495.2499694824219, |
| "completions/min_length": 87.2, |
| "completions/min_terminated_length": 87.2, |
| "entropy": 0.22830554461106659, |
| "epoch": 0.1164079822616408, |
| "frac_reward_zero_std": 0.49375, |
| "grad_norm": 0.29296875, |
| "learning_rate": 9.9791e-06, |
| "loss": -0.0198, |
| "num_tokens": 19689861.0, |
| "reward": 0.33945313543081285, |
| "reward_std": 0.4266162723302841, |
| "rewards/reward_accuracy/mean": 0.24375, |
| "rewards/reward_accuracy/std": 0.4252490371465683, |
| "rewards/reward_format/mean": 0.09570312798023224, |
| "rewards/reward_format/std": 0.014931989088654517, |
| "sampling/importance_sampling_ratio/max": 2.8970901012420653, |
| "sampling/importance_sampling_ratio/mean": 0.7554981231689453, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6467884063720704, |
| "sampling/sampling_logp_difference/mean": 0.012279010005295276, |
| "step": 210, |
| "step_time": 20.485615342622623 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2566.9, |
| "completions/max_terminated_length": 1915.4, |
| "completions/mean_length": 522.63203125, |
| "completions/mean_terminated_length": 508.6317840576172, |
| "completions/min_length": 73.4, |
| "completions/min_terminated_length": 73.4, |
| "entropy": 0.2252771083265543, |
| "epoch": 0.12195121951219512, |
| "frac_reward_zero_std": 0.475, |
| "grad_norm": 0.162109375, |
| "learning_rate": 9.9781e-06, |
| "loss": -0.0048, |
| "num_tokens": 20658638.0, |
| "reward": 0.346250019967556, |
| "reward_std": 0.42528483271598816, |
| "rewards/reward_accuracy/mean": 0.25234375, |
| "rewards/reward_accuracy/std": 0.4229692190885544, |
| "rewards/reward_format/mean": 0.09390625357627869, |
| "rewards/reward_format/std": 0.018736798595637084, |
| "sampling/importance_sampling_ratio/max": 2.766212582588196, |
| "sampling/importance_sampling_ratio/mean": 0.7492749631404877, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6561482131481171, |
| "sampling/sampling_logp_difference/mean": 0.012154272198677063, |
| "step": 220, |
| "step_time": 24.153389699431138 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2451.4, |
| "completions/max_terminated_length": 2242.8, |
| "completions/mean_length": 533.05625, |
| "completions/mean_terminated_length": 525.12783203125, |
| "completions/min_length": 72.2, |
| "completions/min_terminated_length": 72.2, |
| "entropy": 0.2291298707947135, |
| "epoch": 0.12749445676274945, |
| "frac_reward_zero_std": 0.48125, |
| "grad_norm": 0.1328125, |
| "learning_rate": 9.977100000000001e-06, |
| "loss": -0.0029, |
| "num_tokens": 21639742.0, |
| "reward": 0.3436328321695328, |
| "reward_std": 0.4280814528465271, |
| "rewards/reward_accuracy/mean": 0.2484375, |
| "rewards/reward_accuracy/std": 0.42591842710971833, |
| "rewards/reward_format/mean": 0.0951953150331974, |
| "rewards/reward_format/std": 0.015836449339985847, |
| "sampling/importance_sampling_ratio/max": 2.831059718132019, |
| "sampling/importance_sampling_ratio/mean": 0.7256153881549835, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.668182098865509, |
| "sampling/sampling_logp_difference/mean": 0.012273986730724573, |
| "step": 230, |
| "step_time": 22.755798386689275 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 2102.1, |
| "completions/max_terminated_length": 1876.7, |
| "completions/mean_length": 543.3828125, |
| "completions/mean_terminated_length": 539.290966796875, |
| "completions/min_length": 87.1, |
| "completions/min_terminated_length": 87.1, |
| "entropy": 0.22935004653409125, |
| "epoch": 0.13303769401330376, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.2578125, |
| "learning_rate": 9.976100000000001e-06, |
| "loss": -0.0278, |
| "num_tokens": 22638480.0, |
| "reward": 0.32597657293081284, |
| "reward_std": 0.41676448881626127, |
| "rewards/reward_accuracy/mean": 0.23046875, |
| "rewards/reward_accuracy/std": 0.414733549952507, |
| "rewards/reward_format/mean": 0.09550781473517418, |
| "rewards/reward_format/std": 0.015302007086575031, |
| "sampling/importance_sampling_ratio/max": 2.7042718172073363, |
| "sampling/importance_sampling_ratio/mean": 0.7389640867710113, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.717660653591156, |
| "sampling/sampling_logp_difference/mean": 0.012462501134723424, |
| "step": 240, |
| "step_time": 20.18680164567195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2534.9, |
| "completions/max_terminated_length": 1958.9, |
| "completions/mean_length": 526.0140625, |
| "completions/mean_terminated_length": 508.07256469726565, |
| "completions/min_length": 90.2, |
| "completions/min_terminated_length": 90.2, |
| "entropy": 0.22801850251853467, |
| "epoch": 0.1385809312638581, |
| "frac_reward_zero_std": 0.45625, |
| "grad_norm": 0.1796875, |
| "learning_rate": 9.9751e-06, |
| "loss": -0.0044, |
| "num_tokens": 23610194.0, |
| "reward": 0.34960938692092897, |
| "reward_std": 0.43318281769752504, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4308915972709656, |
| "rewards/reward_format/mean": 0.09570312649011611, |
| "rewards/reward_format/std": 0.016069331113249062, |
| "sampling/importance_sampling_ratio/max": 2.765608882904053, |
| "sampling/importance_sampling_ratio/mean": 0.7329097032546997, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.8355767846107482, |
| "sampling/sampling_logp_difference/mean": 0.012590299360454082, |
| "step": 250, |
| "step_time": 24.094800661690535 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 1905.4, |
| "completions/max_terminated_length": 1656.5, |
| "completions/mean_length": 502.42109375, |
| "completions/mean_terminated_length": 498.4580383300781, |
| "completions/min_length": 76.5, |
| "completions/min_terminated_length": 76.5, |
| "entropy": 0.23384508211165667, |
| "epoch": 0.14412416851441243, |
| "frac_reward_zero_std": 0.4375, |
| "grad_norm": 0.2578125, |
| "learning_rate": 9.974100000000002e-06, |
| "loss": -0.0053, |
| "num_tokens": 24550909.0, |
| "reward": 0.34011720567941667, |
| "reward_std": 0.4171771049499512, |
| "rewards/reward_accuracy/mean": 0.2453125, |
| "rewards/reward_accuracy/std": 0.4152470678091049, |
| "rewards/reward_format/mean": 0.09480468928813934, |
| "rewards/reward_format/std": 0.017024051677435637, |
| "sampling/importance_sampling_ratio/max": 2.8384652853012087, |
| "sampling/importance_sampling_ratio/mean": 0.7630635559558868, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7083150386810303, |
| "sampling/sampling_logp_difference/mean": 0.012742383778095246, |
| "step": 260, |
| "step_time": 18.135776404663922 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2733.8, |
| "completions/max_terminated_length": 1972.2, |
| "completions/mean_length": 544.78515625, |
| "completions/mean_terminated_length": 523.0196716308594, |
| "completions/min_length": 88.9, |
| "completions/min_terminated_length": 88.9, |
| "entropy": 0.2244309770874679, |
| "epoch": 0.14966740576496673, |
| "frac_reward_zero_std": 0.475, |
| "grad_norm": 0.21484375, |
| "learning_rate": 9.973100000000001e-06, |
| "loss": -0.0078, |
| "num_tokens": 25544066.0, |
| "reward": 0.35687501132488253, |
| "reward_std": 0.4315678119659424, |
| "rewards/reward_accuracy/mean": 0.2609375, |
| "rewards/reward_accuracy/std": 0.4294391334056854, |
| "rewards/reward_format/mean": 0.09593750312924385, |
| "rewards/reward_format/std": 0.015760267805308103, |
| "sampling/importance_sampling_ratio/max": 2.70762300491333, |
| "sampling/importance_sampling_ratio/mean": 0.7165930390357971, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7067687213420868, |
| "sampling/sampling_logp_difference/mean": 0.011980777699500322, |
| "step": 270, |
| "step_time": 25.565775217534974 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 2130.4, |
| "completions/max_terminated_length": 1791.9, |
| "completions/mean_length": 530.809375, |
| "completions/mean_terminated_length": 524.8954681396484, |
| "completions/min_length": 96.2, |
| "completions/min_terminated_length": 96.2, |
| "entropy": 0.23079944467172026, |
| "epoch": 0.15521064301552107, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.2255859375, |
| "learning_rate": 9.9721e-06, |
| "loss": -0.0053, |
| "num_tokens": 26528262.0, |
| "reward": 0.3230859458446503, |
| "reward_std": 0.4152297377586365, |
| "rewards/reward_accuracy/mean": 0.22734375, |
| "rewards/reward_accuracy/std": 0.4140670448541641, |
| "rewards/reward_format/mean": 0.09574218839406967, |
| "rewards/reward_format/std": 0.015091788768768311, |
| "sampling/importance_sampling_ratio/max": 2.7150286197662354, |
| "sampling/importance_sampling_ratio/mean": 0.7119422852993011, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6720974624156952, |
| "sampling/sampling_logp_difference/mean": 0.012543925363570452, |
| "step": 280, |
| "step_time": 20.070575729943812 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 2262.1, |
| "completions/max_terminated_length": 1914.8, |
| "completions/mean_length": 518.18984375, |
| "completions/mean_terminated_length": 512.1574462890625, |
| "completions/min_length": 98.0, |
| "completions/min_terminated_length": 98.0, |
| "entropy": 0.230003603361547, |
| "epoch": 0.1607538802660754, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.1806640625, |
| "learning_rate": 9.971100000000002e-06, |
| "loss": -0.0131, |
| "num_tokens": 27491841.0, |
| "reward": 0.35136719793081284, |
| "reward_std": 0.4254400610923767, |
| "rewards/reward_accuracy/mean": 0.25625, |
| "rewards/reward_accuracy/std": 0.42321169674396514, |
| "rewards/reward_format/mean": 0.09511718899011612, |
| "rewards/reward_format/std": 0.015806481521576644, |
| "sampling/importance_sampling_ratio/max": 2.7870453596115112, |
| "sampling/importance_sampling_ratio/mean": 0.7226952493190766, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7967423319816589, |
| "sampling/sampling_logp_difference/mean": 0.012381318397819996, |
| "step": 290, |
| "step_time": 21.040557951666415 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2653.2, |
| "completions/max_terminated_length": 2018.7, |
| "completions/mean_length": 511.69140625, |
| "completions/mean_terminated_length": 495.57500610351565, |
| "completions/min_length": 86.7, |
| "completions/min_terminated_length": 86.7, |
| "entropy": 0.23639674875885247, |
| "epoch": 0.1662971175166297, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.224609375, |
| "learning_rate": 9.9701e-06, |
| "loss": -0.0212, |
| "num_tokens": 28448166.0, |
| "reward": 0.329531267285347, |
| "reward_std": 0.41943986117839815, |
| "rewards/reward_accuracy/mean": 0.23359375, |
| "rewards/reward_accuracy/std": 0.41804392635822296, |
| "rewards/reward_format/mean": 0.09593749940395355, |
| "rewards/reward_format/std": 0.01580556146800518, |
| "sampling/importance_sampling_ratio/max": 2.6859883069992065, |
| "sampling/importance_sampling_ratio/mean": 0.7013330519199371, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.6344353199005126, |
| "sampling/sampling_logp_difference/mean": 0.012898232229053974, |
| "step": 300, |
| "step_time": 24.735302259353922 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2695.0, |
| "completions/max_terminated_length": 2223.2, |
| "completions/mean_length": 522.2390625, |
| "completions/mean_terminated_length": 510.16892700195314, |
| "completions/min_length": 85.8, |
| "completions/min_terminated_length": 85.8, |
| "entropy": 0.22403049934655428, |
| "epoch": 0.17184035476718404, |
| "frac_reward_zero_std": 0.43125, |
| "grad_norm": 0.1787109375, |
| "learning_rate": 9.9691e-06, |
| "loss": -0.0075, |
| "num_tokens": 29408432.0, |
| "reward": 0.37640626430511476, |
| "reward_std": 0.44134590923786166, |
| "rewards/reward_accuracy/mean": 0.28125, |
| "rewards/reward_accuracy/std": 0.4388121098279953, |
| "rewards/reward_format/mean": 0.0951562501490116, |
| "rewards/reward_format/std": 0.015351621061563491, |
| "sampling/importance_sampling_ratio/max": 2.6576178789138796, |
| "sampling/importance_sampling_ratio/mean": 0.7529934644699097, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6224551439285279, |
| "sampling/sampling_logp_difference/mean": 0.011854196619242429, |
| "step": 310, |
| "step_time": 25.241768027236684 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 2140.5, |
| "completions/max_terminated_length": 1905.2, |
| "completions/mean_length": 549.6828125, |
| "completions/mean_terminated_length": 545.7215972900391, |
| "completions/min_length": 113.0, |
| "completions/min_terminated_length": 113.0, |
| "entropy": 0.22074419362470507, |
| "epoch": 0.17738359201773837, |
| "frac_reward_zero_std": 0.48125, |
| "grad_norm": 0.2158203125, |
| "learning_rate": 9.9681e-06, |
| "loss": 0.0095, |
| "num_tokens": 30417042.0, |
| "reward": 0.3158984571695328, |
| "reward_std": 0.40523844957351685, |
| "rewards/reward_accuracy/mean": 0.21953125, |
| "rewards/reward_accuracy/std": 0.4033483982086182, |
| "rewards/reward_format/mean": 0.09636719003319741, |
| "rewards/reward_format/std": 0.012644087569788099, |
| "sampling/importance_sampling_ratio/max": 2.778014874458313, |
| "sampling/importance_sampling_ratio/mean": 0.7233938992023468, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6774742603302002, |
| "sampling/sampling_logp_difference/mean": 0.01193384751677513, |
| "step": 320, |
| "step_time": 19.691591003956272 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2958.0, |
| "completions/max_terminated_length": 2105.4, |
| "completions/mean_length": 558.42265625, |
| "completions/mean_terminated_length": 540.7206787109375, |
| "completions/min_length": 97.4, |
| "completions/min_terminated_length": 97.4, |
| "entropy": 0.21981262285262346, |
| "epoch": 0.18292682926829268, |
| "frac_reward_zero_std": 0.41875, |
| "grad_norm": 0.2333984375, |
| "learning_rate": 9.9671e-06, |
| "loss": -0.0116, |
| "num_tokens": 31428991.0, |
| "reward": 0.312421889603138, |
| "reward_std": 0.4121430188417435, |
| "rewards/reward_accuracy/mean": 0.21640625, |
| "rewards/reward_accuracy/std": 0.4101191282272339, |
| "rewards/reward_format/mean": 0.09601562768220902, |
| "rewards/reward_format/std": 0.015373502857983112, |
| "sampling/importance_sampling_ratio/max": 2.825428676605225, |
| "sampling/importance_sampling_ratio/mean": 0.7437629759311676, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6967857837677002, |
| "sampling/sampling_logp_difference/mean": 0.011886996403336524, |
| "step": 330, |
| "step_time": 26.93819950069301 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2406.0, |
| "completions/max_terminated_length": 2005.6, |
| "completions/mean_length": 536.06796875, |
| "completions/mean_terminated_length": 528.2976989746094, |
| "completions/min_length": 85.7, |
| "completions/min_terminated_length": 85.7, |
| "entropy": 0.22054355489090086, |
| "epoch": 0.188470066518847, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.26171875, |
| "learning_rate": 9.966100000000001e-06, |
| "loss": -0.0118, |
| "num_tokens": 32418550.0, |
| "reward": 0.35222657918930056, |
| "reward_std": 0.43409755527973176, |
| "rewards/reward_accuracy/mean": 0.25546875, |
| "rewards/reward_accuracy/std": 0.4327725648880005, |
| "rewards/reward_format/mean": 0.09675781577825546, |
| "rewards/reward_format/std": 0.012879634974524379, |
| "sampling/importance_sampling_ratio/max": 2.772004771232605, |
| "sampling/importance_sampling_ratio/mean": 0.7645464360713958, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7495891928672791, |
| "sampling/sampling_logp_difference/mean": 0.011820383369922638, |
| "step": 340, |
| "step_time": 22.455023988103495 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2236.0, |
| "completions/max_terminated_length": 1978.9, |
| "completions/mean_length": 501.55078125, |
| "completions/mean_terminated_length": 489.5506958007812, |
| "completions/min_length": 93.6, |
| "completions/min_terminated_length": 93.6, |
| "entropy": 0.23104140078648924, |
| "epoch": 0.19401330376940132, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.177734375, |
| "learning_rate": 9.9651e-06, |
| "loss": -0.0201, |
| "num_tokens": 33348783.0, |
| "reward": 0.37507814168930054, |
| "reward_std": 0.44246406853199005, |
| "rewards/reward_accuracy/mean": 0.27734375, |
| "rewards/reward_accuracy/std": 0.44108148813247683, |
| "rewards/reward_format/mean": 0.09773437678813934, |
| "rewards/reward_format/std": 0.011225985456258058, |
| "sampling/importance_sampling_ratio/max": 2.7475876808166504, |
| "sampling/importance_sampling_ratio/mean": 0.7772200167179107, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7617266654968262, |
| "sampling/sampling_logp_difference/mean": 0.01265274714678526, |
| "step": 350, |
| "step_time": 20.941120496531948 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2596.4, |
| "completions/max_terminated_length": 1852.3, |
| "completions/mean_length": 508.7484375, |
| "completions/mean_terminated_length": 494.7001190185547, |
| "completions/min_length": 95.3, |
| "completions/min_terminated_length": 95.3, |
| "entropy": 0.22946543004363776, |
| "epoch": 0.19955654101995565, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.361328125, |
| "learning_rate": 9.964100000000002e-06, |
| "loss": 0.002, |
| "num_tokens": 34298221.0, |
| "reward": 0.3227343887090683, |
| "reward_std": 0.4118516743183136, |
| "rewards/reward_accuracy/mean": 0.22578125, |
| "rewards/reward_accuracy/std": 0.4104402482509613, |
| "rewards/reward_format/mean": 0.09695312604308129, |
| "rewards/reward_format/std": 0.013645543530583382, |
| "sampling/importance_sampling_ratio/max": 2.646984887123108, |
| "sampling/importance_sampling_ratio/mean": 0.7382079780101776, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7618290722370148, |
| "sampling/sampling_logp_difference/mean": 0.012350748479366302, |
| "step": 360, |
| "step_time": 24.038934951182455 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2244.6, |
| "completions/max_terminated_length": 1903.5, |
| "completions/mean_length": 501.54375, |
| "completions/mean_terminated_length": 489.46729431152346, |
| "completions/min_length": 84.1, |
| "completions/min_terminated_length": 84.1, |
| "entropy": 0.21745313201099634, |
| "epoch": 0.20509977827050999, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.16015625, |
| "learning_rate": 9.963100000000001e-06, |
| "loss": -0.0007, |
| "num_tokens": 35239893.0, |
| "reward": 0.37863283306360246, |
| "reward_std": 0.4396579056978226, |
| "rewards/reward_accuracy/mean": 0.28203125, |
| "rewards/reward_accuracy/std": 0.4384962171316147, |
| "rewards/reward_format/mean": 0.09660156294703484, |
| "rewards/reward_format/std": 0.013299217540770769, |
| "sampling/importance_sampling_ratio/max": 2.737509822845459, |
| "sampling/importance_sampling_ratio/mean": 0.7579805672168731, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7347296476364136, |
| "sampling/sampling_logp_difference/mean": 0.012011326849460602, |
| "step": 370, |
| "step_time": 20.912372388085352 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2436.8, |
| "completions/max_terminated_length": 2036.2, |
| "completions/mean_length": 515.9890625, |
| "completions/mean_terminated_length": 506.0520385742187, |
| "completions/min_length": 102.8, |
| "completions/min_terminated_length": 102.8, |
| "entropy": 0.22590704811736942, |
| "epoch": 0.2106430155210643, |
| "frac_reward_zero_std": 0.45, |
| "grad_norm": 0.2373046875, |
| "learning_rate": 9.9621e-06, |
| "loss": 0.0002, |
| "num_tokens": 36213287.0, |
| "reward": 0.30304689705371857, |
| "reward_std": 0.39799932241439817, |
| "rewards/reward_accuracy/mean": 0.20625, |
| "rewards/reward_accuracy/std": 0.39630998820066454, |
| "rewards/reward_format/mean": 0.0967968761920929, |
| "rewards/reward_format/std": 0.012896646745502949, |
| "sampling/importance_sampling_ratio/max": 2.8113519191741942, |
| "sampling/importance_sampling_ratio/mean": 0.7607064485549927, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6604692339897156, |
| "sampling/sampling_logp_difference/mean": 0.012266407907009124, |
| "step": 380, |
| "step_time": 22.56825194763951 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2402.9, |
| "completions/max_terminated_length": 2094.4, |
| "completions/mean_length": 496.34765625, |
| "completions/mean_terminated_length": 484.46498413085936, |
| "completions/min_length": 92.7, |
| "completions/min_terminated_length": 92.7, |
| "entropy": 0.228389014583081, |
| "epoch": 0.21618625277161863, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.1982421875, |
| "learning_rate": 9.961100000000002e-06, |
| "loss": -0.0178, |
| "num_tokens": 37150260.0, |
| "reward": 0.31781251132488253, |
| "reward_std": 0.4133699357509613, |
| "rewards/reward_accuracy/mean": 0.22109375, |
| "rewards/reward_accuracy/std": 0.41235644817352296, |
| "rewards/reward_format/mean": 0.09671875387430191, |
| "rewards/reward_format/std": 0.012513736728578806, |
| "sampling/importance_sampling_ratio/max": 2.725927972793579, |
| "sampling/importance_sampling_ratio/mean": 0.7269497454166413, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7721059083938598, |
| "sampling/sampling_logp_difference/mean": 0.012572839017957449, |
| "step": 390, |
| "step_time": 21.71842490336858 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2840.3, |
| "completions/max_terminated_length": 2175.5, |
| "completions/mean_length": 555.0, |
| "completions/mean_terminated_length": 533.1756622314454, |
| "completions/min_length": 86.3, |
| "completions/min_terminated_length": 86.3, |
| "entropy": 0.22671255487948655, |
| "epoch": 0.22172949002217296, |
| "frac_reward_zero_std": 0.5625, |
| "grad_norm": 0.265625, |
| "learning_rate": 9.9601e-06, |
| "loss": -0.0013, |
| "num_tokens": 38171476.0, |
| "reward": 0.33304689079523087, |
| "reward_std": 0.3815509153530002, |
| "rewards/reward_accuracy/mean": 0.23671875, |
| "rewards/reward_accuracy/std": 0.3783405929803848, |
| "rewards/reward_format/mean": 0.0963281273841858, |
| "rewards/reward_format/std": 0.015542680583894252, |
| "sampling/importance_sampling_ratio/max": 2.6890528917312624, |
| "sampling/importance_sampling_ratio/mean": 0.735101717710495, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7448558568954468, |
| "sampling/sampling_logp_difference/mean": 0.012122452072799206, |
| "step": 400, |
| "step_time": 26.56482650260441 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2544.6, |
| "completions/max_terminated_length": 2203.4, |
| "completions/mean_length": 532.37578125, |
| "completions/mean_terminated_length": 518.3191467285156, |
| "completions/min_length": 86.6, |
| "completions/min_terminated_length": 86.6, |
| "entropy": 0.21303504332900047, |
| "epoch": 0.22727272727272727, |
| "frac_reward_zero_std": 0.46875, |
| "grad_norm": 0.1298828125, |
| "learning_rate": 9.959100000000001e-06, |
| "loss": -0.0123, |
| "num_tokens": 39152565.0, |
| "reward": 0.3412890747189522, |
| "reward_std": 0.4248190730810165, |
| "rewards/reward_accuracy/mean": 0.24453125, |
| "rewards/reward_accuracy/std": 0.4238706588745117, |
| "rewards/reward_format/mean": 0.0967578150331974, |
| "rewards/reward_format/std": 0.013532718969509005, |
| "sampling/importance_sampling_ratio/max": 2.888198494911194, |
| "sampling/importance_sampling_ratio/mean": 0.7659822165966034, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7854882121086121, |
| "sampling/sampling_logp_difference/mean": 0.01155612338334322, |
| "step": 410, |
| "step_time": 23.733121332898737 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2796.0, |
| "completions/max_terminated_length": 2109.1, |
| "completions/mean_length": 586.078125, |
| "completions/mean_terminated_length": 568.4835205078125, |
| "completions/min_length": 107.7, |
| "completions/min_terminated_length": 107.7, |
| "entropy": 0.21583024952560664, |
| "epoch": 0.2328159645232816, |
| "frac_reward_zero_std": 0.40625, |
| "grad_norm": 0.2177734375, |
| "learning_rate": 9.9581e-06, |
| "loss": -0.0083, |
| "num_tokens": 40203433.0, |
| "reward": 0.31406251192092893, |
| "reward_std": 0.40353170335292815, |
| "rewards/reward_accuracy/mean": 0.21875, |
| "rewards/reward_accuracy/std": 0.4013113588094711, |
| "rewards/reward_format/mean": 0.09531250149011612, |
| "rewards/reward_format/std": 0.016169614531099795, |
| "sampling/importance_sampling_ratio/max": 2.8015020847320558, |
| "sampling/importance_sampling_ratio/mean": 0.7296843469142914, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.9027101039886475, |
| "sampling/sampling_logp_difference/mean": 0.011669689510017633, |
| "step": 420, |
| "step_time": 25.93293310967274 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2543.9, |
| "completions/max_terminated_length": 1788.2, |
| "completions/mean_length": 481.6, |
| "completions/mean_terminated_length": 465.28477478027344, |
| "completions/min_length": 85.2, |
| "completions/min_terminated_length": 85.2, |
| "entropy": 0.22163083013147117, |
| "epoch": 0.23835920177383593, |
| "frac_reward_zero_std": 0.4625, |
| "grad_norm": 0.220703125, |
| "learning_rate": 9.9571e-06, |
| "loss": 0.0149, |
| "num_tokens": 41117241.0, |
| "reward": 0.39246095418930055, |
| "reward_std": 0.45547507107257845, |
| "rewards/reward_accuracy/mean": 0.29609375, |
| "rewards/reward_accuracy/std": 0.4541184425354004, |
| "rewards/reward_format/mean": 0.09636718854308128, |
| "rewards/reward_format/std": 0.014589947834610938, |
| "sampling/importance_sampling_ratio/max": 2.8308954000473023, |
| "sampling/importance_sampling_ratio/mean": 0.800515204668045, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6170210003852844, |
| "sampling/sampling_logp_difference/mean": 0.011771329306066036, |
| "step": 430, |
| "step_time": 23.386130075762047 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2358.6, |
| "completions/max_terminated_length": 1953.7, |
| "completions/mean_length": 492.41484375, |
| "completions/mean_terminated_length": 482.361083984375, |
| "completions/min_length": 91.3, |
| "completions/min_terminated_length": 91.3, |
| "entropy": 0.23337912065908312, |
| "epoch": 0.24390243902439024, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.2392578125, |
| "learning_rate": 9.956100000000001e-06, |
| "loss": -0.0068, |
| "num_tokens": 42042924.0, |
| "reward": 0.3709765821695328, |
| "reward_std": 0.434235018491745, |
| "rewards/reward_accuracy/mean": 0.2734375, |
| "rewards/reward_accuracy/std": 0.433160537481308, |
| "rewards/reward_format/mean": 0.09753906354308128, |
| "rewards/reward_format/std": 0.011827559024095536, |
| "sampling/importance_sampling_ratio/max": 2.8563345193862917, |
| "sampling/importance_sampling_ratio/mean": 0.7410735964775086, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6537671327590943, |
| "sampling/sampling_logp_difference/mean": 0.012748954817652702, |
| "step": 440, |
| "step_time": 21.793993984069676 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2255.2, |
| "completions/max_terminated_length": 1966.7, |
| "completions/mean_length": 531.853125, |
| "completions/mean_terminated_length": 520.103271484375, |
| "completions/min_length": 91.2, |
| "completions/min_terminated_length": 91.2, |
| "entropy": 0.2185235286131501, |
| "epoch": 0.24944567627494457, |
| "frac_reward_zero_std": 0.48125, |
| "grad_norm": 0.2578125, |
| "learning_rate": 9.9551e-06, |
| "loss": -0.0182, |
| "num_tokens": 43016568.0, |
| "reward": 0.3721484556794167, |
| "reward_std": 0.4350940495729446, |
| "rewards/reward_accuracy/mean": 0.27578125, |
| "rewards/reward_accuracy/std": 0.4334251582622528, |
| "rewards/reward_format/mean": 0.09636719077825547, |
| "rewards/reward_format/std": 0.014799102209508419, |
| "sampling/importance_sampling_ratio/max": 2.788657522201538, |
| "sampling/importance_sampling_ratio/mean": 0.7750284194946289, |
| "sampling/importance_sampling_ratio/min": 0.0013064678758382797, |
| "sampling/sampling_logp_difference/max": 0.747321081161499, |
| "sampling/sampling_logp_difference/mean": 0.011726300977170468, |
| "step": 450, |
| "step_time": 21.192254751268774 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2702.9, |
| "completions/max_terminated_length": 1870.1, |
| "completions/mean_length": 525.1796875, |
| "completions/mean_terminated_length": 507.2255065917969, |
| "completions/min_length": 94.9, |
| "completions/min_terminated_length": 94.9, |
| "entropy": 0.22623870680108665, |
| "epoch": 0.2549889135254989, |
| "frac_reward_zero_std": 0.45, |
| "grad_norm": 0.25, |
| "learning_rate": 9.954100000000002e-06, |
| "loss": 0.0165, |
| "num_tokens": 43984478.0, |
| "reward": 0.3203906327486038, |
| "reward_std": 0.4107436120510101, |
| "rewards/reward_accuracy/mean": 0.22421875, |
| "rewards/reward_accuracy/std": 0.40966791212558745, |
| "rewards/reward_format/mean": 0.0961718775331974, |
| "rewards/reward_format/std": 0.015184687823057175, |
| "sampling/importance_sampling_ratio/max": 2.861821436882019, |
| "sampling/importance_sampling_ratio/mean": 0.7861462473869324, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7312800645828247, |
| "sampling/sampling_logp_difference/mean": 0.01211005449295044, |
| "step": 460, |
| "step_time": 24.911376668466254 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2791.6, |
| "completions/max_terminated_length": 1896.6, |
| "completions/mean_length": 518.9953125, |
| "completions/mean_terminated_length": 490.86683959960936, |
| "completions/min_length": 94.5, |
| "completions/min_terminated_length": 94.5, |
| "entropy": 0.21905112881213426, |
| "epoch": 0.26053215077605324, |
| "frac_reward_zero_std": 0.51875, |
| "grad_norm": 0.232421875, |
| "learning_rate": 9.953100000000001e-06, |
| "loss": -0.0061, |
| "num_tokens": 44947800.0, |
| "reward": 0.3576171875, |
| "reward_std": 0.4405129015445709, |
| "rewards/reward_accuracy/mean": 0.26171875, |
| "rewards/reward_accuracy/std": 0.4383685886859894, |
| "rewards/reward_format/mean": 0.09589843899011612, |
| "rewards/reward_format/std": 0.01595460968092084, |
| "sampling/importance_sampling_ratio/max": 2.727881073951721, |
| "sampling/importance_sampling_ratio/mean": 0.7596394062042237, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.323678481578827, |
| "sampling/sampling_logp_difference/mean": 0.011852286756038666, |
| "step": 470, |
| "step_time": 25.695770190423353 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2619.3, |
| "completions/max_terminated_length": 2098.8, |
| "completions/mean_length": 547.13359375, |
| "completions/mean_terminated_length": 533.1065612792969, |
| "completions/min_length": 79.1, |
| "completions/min_terminated_length": 79.1, |
| "entropy": 0.2114588800817728, |
| "epoch": 0.2660753880266075, |
| "frac_reward_zero_std": 0.49375, |
| "grad_norm": 0.228515625, |
| "learning_rate": 9.9521e-06, |
| "loss": 0.0002, |
| "num_tokens": 45958011.0, |
| "reward": 0.32378907799720763, |
| "reward_std": 0.4136329501867294, |
| "rewards/reward_accuracy/mean": 0.22734375, |
| "rewards/reward_accuracy/std": 0.41203178763389586, |
| "rewards/reward_format/mean": 0.09644531235098838, |
| "rewards/reward_format/std": 0.014378146827220916, |
| "sampling/importance_sampling_ratio/max": 2.883492136001587, |
| "sampling/importance_sampling_ratio/mean": 0.7402037978172302, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6180720448493957, |
| "sampling/sampling_logp_difference/mean": 0.011421257257461548, |
| "step": 480, |
| "step_time": 24.32624282510951 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2189.8, |
| "completions/max_terminated_length": 1995.6, |
| "completions/mean_length": 479.7484375, |
| "completions/mean_terminated_length": 469.6677551269531, |
| "completions/min_length": 82.0, |
| "completions/min_terminated_length": 82.0, |
| "entropy": 0.21920242113992572, |
| "epoch": 0.27161862527716185, |
| "frac_reward_zero_std": 0.5625, |
| "grad_norm": 0.203125, |
| "learning_rate": 9.951100000000002e-06, |
| "loss": -0.0031, |
| "num_tokens": 46875105.0, |
| "reward": 0.35605470538139344, |
| "reward_std": 0.43057694733142854, |
| "rewards/reward_accuracy/mean": 0.25859375, |
| "rewards/reward_accuracy/std": 0.429737788438797, |
| "rewards/reward_format/mean": 0.09746093899011612, |
| "rewards/reward_format/std": 0.011707901488989592, |
| "sampling/importance_sampling_ratio/max": 2.7715151786804197, |
| "sampling/importance_sampling_ratio/mean": 0.7627084612846374, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6577386140823365, |
| "sampling/sampling_logp_difference/mean": 0.011916977632790805, |
| "step": 490, |
| "step_time": 20.228992110677062 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2415.0, |
| "completions/max_terminated_length": 2105.8, |
| "completions/mean_length": 491.05390625, |
| "completions/mean_terminated_length": 483.0641235351562, |
| "completions/min_length": 86.1, |
| "completions/min_terminated_length": 86.1, |
| "entropy": 0.2072349578142166, |
| "epoch": 0.2771618625277162, |
| "frac_reward_zero_std": 0.525, |
| "grad_norm": 0.224609375, |
| "learning_rate": 9.9501e-06, |
| "loss": -0.0004, |
| "num_tokens": 47800526.0, |
| "reward": 0.3753515839576721, |
| "reward_std": 0.4384240746498108, |
| "rewards/reward_accuracy/mean": 0.278125, |
| "rewards/reward_accuracy/std": 0.4376436024904251, |
| "rewards/reward_format/mean": 0.09722656533122062, |
| "rewards/reward_format/std": 0.012278880923986435, |
| "sampling/importance_sampling_ratio/max": 2.75300452709198, |
| "sampling/importance_sampling_ratio/mean": 0.7974851787090301, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7219130456447601, |
| "sampling/sampling_logp_difference/mean": 0.011304039880633355, |
| "step": 500, |
| "step_time": 22.17811190043576 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2458.4, |
| "completions/max_terminated_length": 1734.6, |
| "completions/mean_length": 478.29453125, |
| "completions/mean_terminated_length": 458.0706756591797, |
| "completions/min_length": 79.7, |
| "completions/min_terminated_length": 79.7, |
| "entropy": 0.23019773522391915, |
| "epoch": 0.2827050997782705, |
| "frac_reward_zero_std": 0.50625, |
| "grad_norm": 0.263671875, |
| "learning_rate": 9.949100000000001e-06, |
| "loss": -0.0163, |
| "num_tokens": 48718183.0, |
| "reward": 0.37878907918930055, |
| "reward_std": 0.44126162230968474, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.4391816079616547, |
| "rewards/reward_format/mean": 0.09597656577825546, |
| "rewards/reward_format/std": 0.015171341598033905, |
| "sampling/importance_sampling_ratio/max": 2.8064987421035767, |
| "sampling/importance_sampling_ratio/mean": 0.7420895576477051, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6329006314277649, |
| "sampling/sampling_logp_difference/mean": 0.012366740312427283, |
| "step": 510, |
| "step_time": 22.930427482817322 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2417.0, |
| "completions/max_terminated_length": 2001.8, |
| "completions/mean_length": 494.62890625, |
| "completions/mean_terminated_length": 484.57403259277345, |
| "completions/min_length": 87.9, |
| "completions/min_terminated_length": 87.9, |
| "entropy": 0.21088731437921523, |
| "epoch": 0.28824833702882485, |
| "frac_reward_zero_std": 0.55625, |
| "grad_norm": 0.1689453125, |
| "learning_rate": 9.9481e-06, |
| "loss": 0.0005, |
| "num_tokens": 49664780.0, |
| "reward": 0.33402345329523087, |
| "reward_std": 0.4164413183927536, |
| "rewards/reward_accuracy/mean": 0.23671875, |
| "rewards/reward_accuracy/std": 0.41503581404685974, |
| "rewards/reward_format/mean": 0.09730468764901161, |
| "rewards/reward_format/std": 0.011575603764504195, |
| "sampling/importance_sampling_ratio/max": 2.7235563516616823, |
| "sampling/importance_sampling_ratio/mean": 0.7726398944854737, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.699539315700531, |
| "sampling/sampling_logp_difference/mean": 0.011464639659970998, |
| "step": 520, |
| "step_time": 22.737110951635987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2268.1, |
| "completions/max_terminated_length": 2002.9, |
| "completions/mean_length": 497.06484375, |
| "completions/mean_terminated_length": 485.16810607910156, |
| "completions/min_length": 97.8, |
| "completions/min_terminated_length": 97.8, |
| "entropy": 0.21165319271385669, |
| "epoch": 0.29379157427937913, |
| "frac_reward_zero_std": 0.55, |
| "grad_norm": 0.10546875, |
| "learning_rate": 9.9471e-06, |
| "loss": -0.0064, |
| "num_tokens": 50602991.0, |
| "reward": 0.39035158306360246, |
| "reward_std": 0.4365664839744568, |
| "rewards/reward_accuracy/mean": 0.29296875, |
| "rewards/reward_accuracy/std": 0.43510702848434446, |
| "rewards/reward_format/mean": 0.09738281443715095, |
| "rewards/reward_format/std": 0.010936710890382529, |
| "sampling/importance_sampling_ratio/max": 2.7835851907730103, |
| "sampling/importance_sampling_ratio/mean": 0.7721957683563232, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6784532368183136, |
| "sampling/sampling_logp_difference/mean": 0.011541504319757223, |
| "step": 530, |
| "step_time": 21.058022207114846 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2725.3, |
| "completions/max_terminated_length": 2039.4, |
| "completions/mean_length": 527.528125, |
| "completions/mean_terminated_length": 509.6644348144531, |
| "completions/min_length": 78.5, |
| "completions/min_terminated_length": 78.5, |
| "entropy": 0.21572315078228713, |
| "epoch": 0.29933481152993346, |
| "frac_reward_zero_std": 0.46875, |
| "grad_norm": 0.2138671875, |
| "learning_rate": 9.946100000000001e-06, |
| "loss": -0.0187, |
| "num_tokens": 51585051.0, |
| "reward": 0.3480859577655792, |
| "reward_std": 0.42963460385799407, |
| "rewards/reward_accuracy/mean": 0.2515625, |
| "rewards/reward_accuracy/std": 0.4283685117959976, |
| "rewards/reward_format/mean": 0.09652344062924385, |
| "rewards/reward_format/std": 0.013789233937859535, |
| "sampling/importance_sampling_ratio/max": 2.751227283477783, |
| "sampling/importance_sampling_ratio/mean": 0.7380363404750824, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7710355043411254, |
| "sampling/sampling_logp_difference/mean": 0.011749415937811137, |
| "step": 540, |
| "step_time": 25.29974924875423 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 2129.2, |
| "completions/max_terminated_length": 1887.1, |
| "completions/mean_length": 513.75, |
| "completions/mean_terminated_length": 507.7985412597656, |
| "completions/min_length": 88.2, |
| "completions/min_terminated_length": 88.2, |
| "entropy": 0.20425767404958606, |
| "epoch": 0.3048780487804878, |
| "frac_reward_zero_std": 0.5375, |
| "grad_norm": 0.376953125, |
| "learning_rate": 9.9451e-06, |
| "loss": 0.0047, |
| "num_tokens": 52543203.0, |
| "reward": 0.3633203268051147, |
| "reward_std": 0.4307068854570389, |
| "rewards/reward_accuracy/mean": 0.265625, |
| "rewards/reward_accuracy/std": 0.42936954498291013, |
| "rewards/reward_format/mean": 0.09769531488418579, |
| "rewards/reward_format/std": 0.010721076419577003, |
| "sampling/importance_sampling_ratio/max": 2.8055014848709106, |
| "sampling/importance_sampling_ratio/mean": 0.8155152320861816, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.8147411227226258, |
| "sampling/sampling_logp_difference/mean": 0.011140500940382481, |
| "step": 550, |
| "step_time": 19.954704904789104 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2514.9, |
| "completions/max_terminated_length": 2053.8, |
| "completions/mean_length": 530.640625, |
| "completions/mean_terminated_length": 512.5816802978516, |
| "completions/min_length": 86.6, |
| "completions/min_terminated_length": 86.6, |
| "entropy": 0.21509129451587797, |
| "epoch": 0.31042128603104213, |
| "frac_reward_zero_std": 0.55625, |
| "grad_norm": 0.1875, |
| "learning_rate": 9.9441e-06, |
| "loss": -0.0189, |
| "num_tokens": 53531631.0, |
| "reward": 0.3347265735268593, |
| "reward_std": 0.4064569532871246, |
| "rewards/reward_accuracy/mean": 0.2375, |
| "rewards/reward_accuracy/std": 0.40499513149261473, |
| "rewards/reward_format/mean": 0.09722656458616256, |
| "rewards/reward_format/std": 0.01208982551470399, |
| "sampling/importance_sampling_ratio/max": 2.7586877822875975, |
| "sampling/importance_sampling_ratio/mean": 0.7591234505176544, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7208037137985229, |
| "sampling/sampling_logp_difference/mean": 0.011431171279400586, |
| "step": 560, |
| "step_time": 23.875041725300253 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1813.0, |
| "completions/max_terminated_length": 1813.0, |
| "completions/mean_length": 456.7125, |
| "completions/mean_terminated_length": 456.7125, |
| "completions/min_length": 83.0, |
| "completions/min_terminated_length": 83.0, |
| "entropy": 0.21012900257483125, |
| "epoch": 0.31596452328159647, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.212890625, |
| "learning_rate": 9.943100000000001e-06, |
| "loss": -0.0129, |
| "num_tokens": 54413207.0, |
| "reward": 0.3673828303813934, |
| "reward_std": 0.4384054273366928, |
| "rewards/reward_accuracy/mean": 0.26875, |
| "rewards/reward_accuracy/std": 0.4379043489694595, |
| "rewards/reward_format/mean": 0.09863281399011611, |
| "rewards/reward_format/std": 0.008129600901156664, |
| "sampling/importance_sampling_ratio/max": 2.816672372817993, |
| "sampling/importance_sampling_ratio/mean": 0.7667007565498352, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6956488847732544, |
| "sampling/sampling_logp_difference/mean": 0.011437455099076033, |
| "step": 570, |
| "step_time": 17.28005353892222 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2262.2, |
| "completions/max_terminated_length": 1940.1, |
| "completions/mean_length": 508.33359375, |
| "completions/mean_terminated_length": 500.4394561767578, |
| "completions/min_length": 75.1, |
| "completions/min_terminated_length": 75.1, |
| "entropy": 0.20385388899594545, |
| "epoch": 0.3215077605321508, |
| "frac_reward_zero_std": 0.5625, |
| "grad_norm": 0.1806640625, |
| "learning_rate": 9.942100000000001e-06, |
| "loss": -0.01, |
| "num_tokens": 55375090.0, |
| "reward": 0.33457032442092893, |
| "reward_std": 0.4220437169075012, |
| "rewards/reward_accuracy/mean": 0.23671875, |
| "rewards/reward_accuracy/std": 0.4214779853820801, |
| "rewards/reward_format/mean": 0.09785156399011612, |
| "rewards/reward_format/std": 0.011332144681364297, |
| "sampling/importance_sampling_ratio/max": 2.6732457876205444, |
| "sampling/importance_sampling_ratio/mean": 0.7796825408935547, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6735784769058227, |
| "sampling/sampling_logp_difference/mean": 0.010962016228586436, |
| "step": 580, |
| "step_time": 21.35113308932632 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2504.1, |
| "completions/max_terminated_length": 1993.5, |
| "completions/mean_length": 498.6265625, |
| "completions/mean_terminated_length": 482.6277099609375, |
| "completions/min_length": 89.5, |
| "completions/min_terminated_length": 89.5, |
| "entropy": 0.20437408294528722, |
| "epoch": 0.3270509977827051, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.203125, |
| "learning_rate": 9.941100000000002e-06, |
| "loss": -0.0181, |
| "num_tokens": 56318580.0, |
| "reward": 0.35699220895767214, |
| "reward_std": 0.4333167433738708, |
| "rewards/reward_accuracy/mean": 0.259375, |
| "rewards/reward_accuracy/std": 0.4329672873020172, |
| "rewards/reward_format/mean": 0.09761718958616257, |
| "rewards/reward_format/std": 0.011852531833574176, |
| "sampling/importance_sampling_ratio/max": 2.7547749280929565, |
| "sampling/importance_sampling_ratio/mean": 0.7766368806362152, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6721030712127686, |
| "sampling/sampling_logp_difference/mean": 0.01107011791318655, |
| "step": 590, |
| "step_time": 23.73060810631141 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00234375, |
| "completions/max_length": 1816.6, |
| "completions/max_terminated_length": 1702.0, |
| "completions/mean_length": 464.5796875, |
| "completions/mean_terminated_length": 458.5050506591797, |
| "completions/min_length": 91.0, |
| "completions/min_terminated_length": 91.0, |
| "entropy": 0.20848135529085993, |
| "epoch": 0.3325942350332594, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.1455078125, |
| "learning_rate": 9.9401e-06, |
| "loss": 0.0121, |
| "num_tokens": 57209682.0, |
| "reward": 0.35566407442092896, |
| "reward_std": 0.43313867449760435, |
| "rewards/reward_accuracy/mean": 0.2578125, |
| "rewards/reward_accuracy/std": 0.43270987272262573, |
| "rewards/reward_format/mean": 0.09785156473517417, |
| "rewards/reward_format/std": 0.01029010619968176, |
| "sampling/importance_sampling_ratio/max": 2.8671164751052856, |
| "sampling/importance_sampling_ratio/mean": 0.7884063363075257, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6488892793655395, |
| "sampling/sampling_logp_difference/mean": 0.011255558673292398, |
| "step": 600, |
| "step_time": 17.22526674247347 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2595.0, |
| "completions/max_terminated_length": 2198.8, |
| "completions/mean_length": 509.43828125, |
| "completions/mean_terminated_length": 499.53643493652345, |
| "completions/min_length": 88.6, |
| "completions/min_terminated_length": 88.6, |
| "entropy": 0.20392611464485527, |
| "epoch": 0.33813747228381374, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.298828125, |
| "learning_rate": 9.939100000000001e-06, |
| "loss": -0.0154, |
| "num_tokens": 58162883.0, |
| "reward": 0.32750001847743987, |
| "reward_std": 0.41948417127132415, |
| "rewards/reward_accuracy/mean": 0.2296875, |
| "rewards/reward_accuracy/std": 0.4183623373508453, |
| "rewards/reward_format/mean": 0.09781250208616257, |
| "rewards/reward_format/std": 0.010693395137786865, |
| "sampling/importance_sampling_ratio/max": 2.7836442947387696, |
| "sampling/importance_sampling_ratio/mean": 0.7830377399921418, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.0012446761131286, |
| "sampling/sampling_logp_difference/mean": 0.011127515416592359, |
| "step": 610, |
| "step_time": 23.474077865667642 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2440.4, |
| "completions/max_terminated_length": 1830.8, |
| "completions/mean_length": 482.78359375, |
| "completions/mean_terminated_length": 472.70257263183595, |
| "completions/min_length": 86.4, |
| "completions/min_terminated_length": 86.4, |
| "entropy": 0.2090471311006695, |
| "epoch": 0.3436807095343681, |
| "frac_reward_zero_std": 0.45625, |
| "grad_norm": 0.2451171875, |
| "learning_rate": 9.9381e-06, |
| "loss": -0.0076, |
| "num_tokens": 59080590.0, |
| "reward": 0.39105470925569535, |
| "reward_std": 0.4452354371547699, |
| "rewards/reward_accuracy/mean": 0.29375, |
| "rewards/reward_accuracy/std": 0.4441016525030136, |
| "rewards/reward_format/mean": 0.09730469062924385, |
| "rewards/reward_format/std": 0.010986986570060253, |
| "sampling/importance_sampling_ratio/max": 2.725508689880371, |
| "sampling/importance_sampling_ratio/mean": 0.7691301286220551, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6603422999382019, |
| "sampling/sampling_logp_difference/mean": 0.011403250228613614, |
| "step": 620, |
| "step_time": 22.87826501368545 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2329.5, |
| "completions/max_terminated_length": 1818.5, |
| "completions/mean_length": 494.4640625, |
| "completions/mean_terminated_length": 474.28251953125, |
| "completions/min_length": 91.0, |
| "completions/min_terminated_length": 91.0, |
| "entropy": 0.20600055465474726, |
| "epoch": 0.3492239467849224, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.1591796875, |
| "learning_rate": 9.9371e-06, |
| "loss": -0.0071, |
| "num_tokens": 60011736.0, |
| "reward": 0.3588672041893005, |
| "reward_std": 0.43022640645503996, |
| "rewards/reward_accuracy/mean": 0.2625, |
| "rewards/reward_accuracy/std": 0.42797584235668185, |
| "rewards/reward_format/mean": 0.09636719003319741, |
| "rewards/reward_format/std": 0.014149570185691119, |
| "sampling/importance_sampling_ratio/max": 2.823225736618042, |
| "sampling/importance_sampling_ratio/mean": 0.7791207075119019, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7144589006900788, |
| "sampling/sampling_logp_difference/mean": 0.010979413613677024, |
| "step": 630, |
| "step_time": 22.19919598600827 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2400.4, |
| "completions/max_terminated_length": 2063.0, |
| "completions/mean_length": 485.5015625, |
| "completions/mean_terminated_length": 463.29762573242186, |
| "completions/min_length": 76.1, |
| "completions/min_terminated_length": 76.1, |
| "entropy": 0.2053542592562735, |
| "epoch": 0.35476718403547675, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.111328125, |
| "learning_rate": 9.936100000000001e-06, |
| "loss": -0.0044, |
| "num_tokens": 60944098.0, |
| "reward": 0.37023439407348635, |
| "reward_std": 0.43288091421127317, |
| "rewards/reward_accuracy/mean": 0.2734375, |
| "rewards/reward_accuracy/std": 0.4317664593458176, |
| "rewards/reward_format/mean": 0.09679687470197677, |
| "rewards/reward_format/std": 0.01390917794778943, |
| "sampling/importance_sampling_ratio/max": 2.77753221988678, |
| "sampling/importance_sampling_ratio/mean": 0.8123395144939423, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6826550722122192, |
| "sampling/sampling_logp_difference/mean": 0.010914453957229852, |
| "step": 640, |
| "step_time": 22.748443353595214 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2404.8, |
| "completions/max_terminated_length": 1846.0, |
| "completions/mean_length": 558.28359375, |
| "completions/mean_terminated_length": 538.9341735839844, |
| "completions/min_length": 104.6, |
| "completions/min_terminated_length": 104.6, |
| "entropy": 0.1960991204250604, |
| "epoch": 0.360310421286031, |
| "frac_reward_zero_std": 0.55625, |
| "grad_norm": 0.16796875, |
| "learning_rate": 9.9351e-06, |
| "loss": -0.0269, |
| "num_tokens": 61969877.0, |
| "reward": 0.29914063811302183, |
| "reward_std": 0.3941082388162613, |
| "rewards/reward_accuracy/mean": 0.20234375, |
| "rewards/reward_accuracy/std": 0.393171963095665, |
| "rewards/reward_format/mean": 0.09679687693715096, |
| "rewards/reward_format/std": 0.01283568823710084, |
| "sampling/importance_sampling_ratio/max": 2.7895674228668215, |
| "sampling/importance_sampling_ratio/mean": 0.7732262134552002, |
| "sampling/importance_sampling_ratio/min": 0.0012030726298689841, |
| "sampling/sampling_logp_difference/max": 0.6658833444118499, |
| "sampling/sampling_logp_difference/mean": 0.010545879974961281, |
| "step": 650, |
| "step_time": 22.67471098760143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 2023.9, |
| "completions/max_terminated_length": 1923.7, |
| "completions/mean_length": 483.34765625, |
| "completions/mean_terminated_length": 479.29588012695314, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.2068402892909944, |
| "epoch": 0.36585365853658536, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.1669921875, |
| "learning_rate": 9.9341e-06, |
| "loss": -0.0081, |
| "num_tokens": 62883658.0, |
| "reward": 0.34746095538139343, |
| "reward_std": 0.42729735374450684, |
| "rewards/reward_accuracy/mean": 0.25, |
| "rewards/reward_accuracy/std": 0.42662867307662966, |
| "rewards/reward_format/mean": 0.09746093899011612, |
| "rewards/reward_format/std": 0.011175514059141278, |
| "sampling/importance_sampling_ratio/max": 2.8130202293395996, |
| "sampling/importance_sampling_ratio/mean": 0.788328868150711, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7859536528587341, |
| "sampling/sampling_logp_difference/mean": 0.01128133200109005, |
| "step": 660, |
| "step_time": 18.799795890972018 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2414.2, |
| "completions/max_terminated_length": 1983.5, |
| "completions/mean_length": 523.98203125, |
| "completions/mean_terminated_length": 512.0448760986328, |
| "completions/min_length": 94.5, |
| "completions/min_terminated_length": 94.5, |
| "entropy": 0.20366938626393677, |
| "epoch": 0.3713968957871397, |
| "frac_reward_zero_std": 0.51875, |
| "grad_norm": 0.2373046875, |
| "learning_rate": 9.933100000000002e-06, |
| "loss": -0.0173, |
| "num_tokens": 63857339.0, |
| "reward": 0.32171875685453416, |
| "reward_std": 0.41015804708004, |
| "rewards/reward_accuracy/mean": 0.22421875, |
| "rewards/reward_accuracy/std": 0.4094666033983231, |
| "rewards/reward_format/mean": 0.09750000014901161, |
| "rewards/reward_format/std": 0.01239622849971056, |
| "sampling/importance_sampling_ratio/max": 2.6962289094924925, |
| "sampling/importance_sampling_ratio/mean": 0.7658665120601654, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6997470319271087, |
| "sampling/sampling_logp_difference/mean": 0.010850257147103548, |
| "step": 670, |
| "step_time": 22.538751268945635 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2303.4, |
| "completions/max_terminated_length": 1988.6, |
| "completions/mean_length": 510.3078125, |
| "completions/mean_terminated_length": 498.30492248535154, |
| "completions/min_length": 80.7, |
| "completions/min_terminated_length": 80.7, |
| "entropy": 0.203141900151968, |
| "epoch": 0.376940133037694, |
| "frac_reward_zero_std": 0.5375, |
| "grad_norm": 0.244140625, |
| "learning_rate": 9.932100000000001e-06, |
| "loss": -0.0043, |
| "num_tokens": 64808053.0, |
| "reward": 0.36117188930511473, |
| "reward_std": 0.4297789543867111, |
| "rewards/reward_accuracy/mean": 0.26328125, |
| "rewards/reward_accuracy/std": 0.42863228023052213, |
| "rewards/reward_format/mean": 0.09789062589406967, |
| "rewards/reward_format/std": 0.011106384126469493, |
| "sampling/importance_sampling_ratio/max": 2.7784851789474487, |
| "sampling/importance_sampling_ratio/mean": 0.7989328145980835, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7461124539375306, |
| "sampling/sampling_logp_difference/mean": 0.010861350782215595, |
| "step": 680, |
| "step_time": 21.608661907166244 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2060.1, |
| "completions/max_terminated_length": 1593.9, |
| "completions/mean_length": 478.81640625, |
| "completions/mean_terminated_length": 470.6123321533203, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.1978349108248949, |
| "epoch": 0.38248337028824836, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.1826171875, |
| "learning_rate": 9.9311e-06, |
| "loss": -0.0258, |
| "num_tokens": 65725642.0, |
| "reward": 0.356210957467556, |
| "reward_std": 0.4291324555873871, |
| "rewards/reward_accuracy/mean": 0.2578125, |
| "rewards/reward_accuracy/std": 0.4285455524921417, |
| "rewards/reward_format/mean": 0.09839843809604645, |
| "rewards/reward_format/std": 0.009406374022364616, |
| "sampling/importance_sampling_ratio/max": 2.8692076206207275, |
| "sampling/importance_sampling_ratio/mean": 0.8288196146488189, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7097841918468475, |
| "sampling/sampling_logp_difference/mean": 0.010841783694922924, |
| "step": 690, |
| "step_time": 19.589108880702405 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2093.9, |
| "completions/max_terminated_length": 1664.5, |
| "completions/mean_length": 454.59921875, |
| "completions/mean_terminated_length": 444.36956787109375, |
| "completions/min_length": 82.7, |
| "completions/min_terminated_length": 82.7, |
| "entropy": 0.21336429798975587, |
| "epoch": 0.38802660753880264, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.279296875, |
| "learning_rate": 9.9301e-06, |
| "loss": 0.019, |
| "num_tokens": 66612137.0, |
| "reward": 0.37421876192092896, |
| "reward_std": 0.44521689116954805, |
| "rewards/reward_accuracy/mean": 0.27578125, |
| "rewards/reward_accuracy/std": 0.4445360600948334, |
| "rewards/reward_format/mean": 0.09843750149011612, |
| "rewards/reward_format/std": 0.00808858759701252, |
| "sampling/importance_sampling_ratio/max": 2.810273003578186, |
| "sampling/importance_sampling_ratio/mean": 0.8214177906513214, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6107546985149384, |
| "sampling/sampling_logp_difference/mean": 0.011412415746599435, |
| "step": 700, |
| "step_time": 19.720249932492152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2499.1, |
| "completions/max_terminated_length": 1869.4, |
| "completions/mean_length": 502.20625, |
| "completions/mean_terminated_length": 487.98899230957034, |
| "completions/min_length": 92.4, |
| "completions/min_terminated_length": 92.4, |
| "entropy": 0.20591012826189398, |
| "epoch": 0.39356984478935697, |
| "frac_reward_zero_std": 0.525, |
| "grad_norm": 0.2392578125, |
| "learning_rate": 9.929100000000001e-06, |
| "loss": -0.0111, |
| "num_tokens": 67561513.0, |
| "reward": 0.3256640747189522, |
| "reward_std": 0.41438646614551544, |
| "rewards/reward_accuracy/mean": 0.22890625, |
| "rewards/reward_accuracy/std": 0.4137218236923218, |
| "rewards/reward_format/mean": 0.09675781428813934, |
| "rewards/reward_format/std": 0.013273235503584146, |
| "sampling/importance_sampling_ratio/max": 2.8192128419876097, |
| "sampling/importance_sampling_ratio/mean": 0.8113150596618652, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7383062362670898, |
| "sampling/sampling_logp_difference/mean": 0.01112342467531562, |
| "step": 710, |
| "step_time": 23.724708492495118 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2553.5, |
| "completions/max_terminated_length": 1641.7, |
| "completions/mean_length": 484.15234375, |
| "completions/mean_terminated_length": 461.9533935546875, |
| "completions/min_length": 76.2, |
| "completions/min_terminated_length": 76.2, |
| "entropy": 0.20679257493466138, |
| "epoch": 0.3991130820399113, |
| "frac_reward_zero_std": 0.45625, |
| "grad_norm": 0.19140625, |
| "learning_rate": 9.9281e-06, |
| "loss": -0.0151, |
| "num_tokens": 68489268.0, |
| "reward": 0.37359377145767214, |
| "reward_std": 0.4438827395439148, |
| "rewards/reward_accuracy/mean": 0.27578125, |
| "rewards/reward_accuracy/std": 0.4425703674554825, |
| "rewards/reward_format/mean": 0.09781250357627869, |
| "rewards/reward_format/std": 0.011628437414765358, |
| "sampling/importance_sampling_ratio/max": 2.8022235870361327, |
| "sampling/importance_sampling_ratio/mean": 0.8033082067966462, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6047082901000976, |
| "sampling/sampling_logp_difference/mean": 0.011286563426256179, |
| "step": 720, |
| "step_time": 23.74397625476122 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2571.6, |
| "completions/max_terminated_length": 1811.6, |
| "completions/mean_length": 480.0609375, |
| "completions/mean_terminated_length": 459.6414733886719, |
| "completions/min_length": 79.0, |
| "completions/min_terminated_length": 79.0, |
| "entropy": 0.2014916862361133, |
| "epoch": 0.40465631929046564, |
| "frac_reward_zero_std": 0.5375, |
| "grad_norm": 0.1474609375, |
| "learning_rate": 9.9271e-06, |
| "loss": -0.0118, |
| "num_tokens": 69404194.0, |
| "reward": 0.4255078285932541, |
| "reward_std": 0.46471441984176637, |
| "rewards/reward_accuracy/mean": 0.328125, |
| "rewards/reward_accuracy/std": 0.4635661572217941, |
| "rewards/reward_format/mean": 0.09738281443715095, |
| "rewards/reward_format/std": 0.012711133155971766, |
| "sampling/importance_sampling_ratio/max": 2.7141573429107666, |
| "sampling/importance_sampling_ratio/mean": 0.7976269662380219, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.8307986378669738, |
| "sampling/sampling_logp_difference/mean": 0.010670965071767569, |
| "step": 730, |
| "step_time": 24.203037958219646 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2508.6, |
| "completions/max_terminated_length": 2008.0, |
| "completions/mean_length": 485.1046875, |
| "completions/mean_terminated_length": 472.9481231689453, |
| "completions/min_length": 79.1, |
| "completions/min_terminated_length": 79.1, |
| "entropy": 0.20387705815955998, |
| "epoch": 0.41019955654101997, |
| "frac_reward_zero_std": 0.475, |
| "grad_norm": 0.2275390625, |
| "learning_rate": 9.926100000000001e-06, |
| "loss": -0.0051, |
| "num_tokens": 70326008.0, |
| "reward": 0.36093751788139344, |
| "reward_std": 0.4285079509019852, |
| "rewards/reward_accuracy/mean": 0.26328125, |
| "rewards/reward_accuracy/std": 0.42708416283130646, |
| "rewards/reward_format/mean": 0.09765625298023224, |
| "rewards/reward_format/std": 0.010984869394451379, |
| "sampling/importance_sampling_ratio/max": 2.875459623336792, |
| "sampling/importance_sampling_ratio/mean": 0.800823163986206, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.791225004196167, |
| "sampling/sampling_logp_difference/mean": 0.010965394787490368, |
| "step": 740, |
| "step_time": 23.3568691002205 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2492.1, |
| "completions/max_terminated_length": 1866.1, |
| "completions/mean_length": 490.36953125, |
| "completions/mean_terminated_length": 478.29013671875, |
| "completions/min_length": 97.4, |
| "completions/min_terminated_length": 97.4, |
| "entropy": 0.1945117291994393, |
| "epoch": 0.4157427937915743, |
| "frac_reward_zero_std": 0.55625, |
| "grad_norm": 0.158203125, |
| "learning_rate": 9.925100000000001e-06, |
| "loss": -0.0092, |
| "num_tokens": 71248769.0, |
| "reward": 0.37832031548023226, |
| "reward_std": 0.4352532595396042, |
| "rewards/reward_accuracy/mean": 0.28125, |
| "rewards/reward_accuracy/std": 0.43357751071453093, |
| "rewards/reward_format/mean": 0.0970703125, |
| "rewards/reward_format/std": 0.012510230205953122, |
| "sampling/importance_sampling_ratio/max": 2.656781268119812, |
| "sampling/importance_sampling_ratio/mean": 0.7971257567405701, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6647083163261414, |
| "sampling/sampling_logp_difference/mean": 0.01024281671270728, |
| "step": 750, |
| "step_time": 23.271028404356912 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2360.1, |
| "completions/max_terminated_length": 2018.0, |
| "completions/mean_length": 509.28125, |
| "completions/mean_terminated_length": 491.13682250976564, |
| "completions/min_length": 96.0, |
| "completions/min_terminated_length": 96.0, |
| "entropy": 0.1994618834927678, |
| "epoch": 0.4212860310421286, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.1416015625, |
| "learning_rate": 9.9241e-06, |
| "loss": 0.0046, |
| "num_tokens": 72205753.0, |
| "reward": 0.32257813960313797, |
| "reward_std": 0.4147211581468582, |
| "rewards/reward_accuracy/mean": 0.225, |
| "rewards/reward_accuracy/std": 0.4139979213476181, |
| "rewards/reward_format/mean": 0.0975781261920929, |
| "rewards/reward_format/std": 0.010867161490023137, |
| "sampling/importance_sampling_ratio/max": 2.796861982345581, |
| "sampling/importance_sampling_ratio/mean": 0.8252683699131012, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7058388948440552, |
| "sampling/sampling_logp_difference/mean": 0.010627043899148703, |
| "step": 760, |
| "step_time": 22.183511776011436 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2663.3, |
| "completions/max_terminated_length": 1897.2, |
| "completions/mean_length": 540.1453125, |
| "completions/mean_terminated_length": 522.4401947021485, |
| "completions/min_length": 97.0, |
| "completions/min_terminated_length": 97.0, |
| "entropy": 0.19750828817486762, |
| "epoch": 0.4268292682926829, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.2255859375, |
| "learning_rate": 9.923100000000002e-06, |
| "loss": 0.0027, |
| "num_tokens": 73195211.0, |
| "reward": 0.34550783336162566, |
| "reward_std": 0.41105996966362, |
| "rewards/reward_accuracy/mean": 0.24765625, |
| "rewards/reward_accuracy/std": 0.4097494214773178, |
| "rewards/reward_format/mean": 0.09785156548023224, |
| "rewards/reward_format/std": 0.01169180078431964, |
| "sampling/importance_sampling_ratio/max": 2.7334346532821656, |
| "sampling/importance_sampling_ratio/mean": 0.757252287864685, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7015691041946411, |
| "sampling/sampling_logp_difference/mean": 0.010581954102963208, |
| "step": 770, |
| "step_time": 24.412712253443896 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2564.9, |
| "completions/max_terminated_length": 2034.3, |
| "completions/mean_length": 502.1, |
| "completions/mean_terminated_length": 490.01535949707034, |
| "completions/min_length": 81.2, |
| "completions/min_terminated_length": 81.2, |
| "entropy": 0.19576062066480518, |
| "epoch": 0.43237250554323725, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.21484375, |
| "learning_rate": 9.922100000000001e-06, |
| "loss": -0.0028, |
| "num_tokens": 74137467.0, |
| "reward": 0.3517578333616257, |
| "reward_std": 0.43595556914806366, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4345319330692291, |
| "rewards/reward_format/mean": 0.09785156473517417, |
| "rewards/reward_format/std": 0.010712989792227744, |
| "sampling/importance_sampling_ratio/max": 2.753013563156128, |
| "sampling/importance_sampling_ratio/mean": 0.7845089435577393, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6141059815883636, |
| "sampling/sampling_logp_difference/mean": 0.010378810483962298, |
| "step": 780, |
| "step_time": 24.09986651148647 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2379.1, |
| "completions/max_terminated_length": 1942.6, |
| "completions/mean_length": 482.34765625, |
| "completions/mean_terminated_length": 472.24379577636716, |
| "completions/min_length": 87.4, |
| "completions/min_terminated_length": 87.4, |
| "entropy": 0.20661399783566595, |
| "epoch": 0.4379157427937916, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.267578125, |
| "learning_rate": 9.9211e-06, |
| "loss": -0.0102, |
| "num_tokens": 75056872.0, |
| "reward": 0.33023439049720765, |
| "reward_std": 0.4142072558403015, |
| "rewards/reward_accuracy/mean": 0.23203125, |
| "rewards/reward_accuracy/std": 0.4139157384634018, |
| "rewards/reward_format/mean": 0.09820312857627869, |
| "rewards/reward_format/std": 0.009844285761937499, |
| "sampling/importance_sampling_ratio/max": 2.7141240358352663, |
| "sampling/importance_sampling_ratio/mean": 0.7707675278186799, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6353932917118073, |
| "sampling/sampling_logp_difference/mean": 0.011057981662452221, |
| "step": 790, |
| "step_time": 22.189873471250756 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2837.8, |
| "completions/max_terminated_length": 2085.8, |
| "completions/mean_length": 520.8, |
| "completions/mean_terminated_length": 500.70847778320314, |
| "completions/min_length": 85.7, |
| "completions/min_terminated_length": 85.7, |
| "entropy": 0.19708600463345646, |
| "epoch": 0.4434589800443459, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.1904296875, |
| "learning_rate": 9.9201e-06, |
| "loss": -0.0177, |
| "num_tokens": 76026944.0, |
| "reward": 0.35917970538139343, |
| "reward_std": 0.4260776609182358, |
| "rewards/reward_accuracy/mean": 0.26171875, |
| "rewards/reward_accuracy/std": 0.42499605715274813, |
| "rewards/reward_format/mean": 0.09746093899011612, |
| "rewards/reward_format/std": 0.012769796140491962, |
| "sampling/importance_sampling_ratio/max": 2.771568512916565, |
| "sampling/importance_sampling_ratio/mean": 0.8070327699184418, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6958881616592407, |
| "sampling/sampling_logp_difference/mean": 0.010527356714010238, |
| "step": 800, |
| "step_time": 26.025093877734616 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2692.1, |
| "completions/max_terminated_length": 2083.4, |
| "completions/mean_length": 542.52578125, |
| "completions/mean_terminated_length": 522.7526062011718, |
| "completions/min_length": 83.2, |
| "completions/min_terminated_length": 83.2, |
| "entropy": 0.19059578012675046, |
| "epoch": 0.4490022172949002, |
| "frac_reward_zero_std": 0.56875, |
| "grad_norm": 0.18359375, |
| "learning_rate": 9.9191e-06, |
| "loss": -0.0276, |
| "num_tokens": 77022833.0, |
| "reward": 0.3884765774011612, |
| "reward_std": 0.43896985948085787, |
| "rewards/reward_accuracy/mean": 0.290625, |
| "rewards/reward_accuracy/std": 0.4380782157182693, |
| "rewards/reward_format/mean": 0.0978515625, |
| "rewards/reward_format/std": 0.010259101446717978, |
| "sampling/importance_sampling_ratio/max": 2.773747515678406, |
| "sampling/importance_sampling_ratio/mean": 0.7861079514026642, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6419491648674012, |
| "sampling/sampling_logp_difference/mean": 0.010310206934809685, |
| "step": 810, |
| "step_time": 25.508965820912273 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2850.1, |
| "completions/max_terminated_length": 1953.6, |
| "completions/mean_length": 545.63359375, |
| "completions/mean_terminated_length": 525.6539947509766, |
| "completions/min_length": 98.8, |
| "completions/min_terminated_length": 98.8, |
| "entropy": 0.19333943398669362, |
| "epoch": 0.45454545454545453, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.3046875, |
| "learning_rate": 9.9181e-06, |
| "loss": -0.0002, |
| "num_tokens": 78031012.0, |
| "reward": 0.35402344465255736, |
| "reward_std": 0.42295392155647277, |
| "rewards/reward_accuracy/mean": 0.25625, |
| "rewards/reward_accuracy/std": 0.4218521326780319, |
| "rewards/reward_format/mean": 0.09777343794703483, |
| "rewards/reward_format/std": 0.012026132922619582, |
| "sampling/importance_sampling_ratio/max": 2.719328999519348, |
| "sampling/importance_sampling_ratio/mean": 0.7902082562446594, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6054224073886871, |
| "sampling/sampling_logp_difference/mean": 0.010355181712657213, |
| "step": 820, |
| "step_time": 26.28215200561099 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2425.2, |
| "completions/max_terminated_length": 1967.4, |
| "completions/mean_length": 489.51484375, |
| "completions/mean_terminated_length": 473.35167236328124, |
| "completions/min_length": 88.1, |
| "completions/min_terminated_length": 88.1, |
| "entropy": 0.19792858017608522, |
| "epoch": 0.46008869179600886, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.25, |
| "learning_rate": 9.9171e-06, |
| "loss": -0.0017, |
| "num_tokens": 78960639.0, |
| "reward": 0.33226563930511477, |
| "reward_std": 0.41683932244777677, |
| "rewards/reward_accuracy/mean": 0.234375, |
| "rewards/reward_accuracy/std": 0.4159123569726944, |
| "rewards/reward_format/mean": 0.09789062887430192, |
| "rewards/reward_format/std": 0.011312332656234502, |
| "sampling/importance_sampling_ratio/max": 2.6689173698425295, |
| "sampling/importance_sampling_ratio/mean": 0.8113642454147338, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6565281331539154, |
| "sampling/sampling_logp_difference/mean": 0.010662268474698066, |
| "step": 830, |
| "step_time": 22.679021937819197 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2601.2, |
| "completions/max_terminated_length": 1918.1, |
| "completions/mean_length": 529.84296875, |
| "completions/mean_terminated_length": 511.94811096191404, |
| "completions/min_length": 74.7, |
| "completions/min_terminated_length": 74.7, |
| "entropy": 0.19611125038936733, |
| "epoch": 0.4656319290465632, |
| "frac_reward_zero_std": 0.55625, |
| "grad_norm": 0.1376953125, |
| "learning_rate": 9.916100000000002e-06, |
| "loss": -0.0051, |
| "num_tokens": 79934726.0, |
| "reward": 0.37695313394069674, |
| "reward_std": 0.44314174354076385, |
| "rewards/reward_accuracy/mean": 0.27890625, |
| "rewards/reward_accuracy/std": 0.44267522990703584, |
| "rewards/reward_format/mean": 0.09804687574505806, |
| "rewards/reward_format/std": 0.011247582826763391, |
| "sampling/importance_sampling_ratio/max": 2.779673361778259, |
| "sampling/importance_sampling_ratio/mean": 0.7712730944156647, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7486573040485383, |
| "sampling/sampling_logp_difference/mean": 0.010390200838446616, |
| "step": 840, |
| "step_time": 24.097158142924307 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2456.9, |
| "completions/max_terminated_length": 1857.7, |
| "completions/mean_length": 491.321875, |
| "completions/mean_terminated_length": 477.1853546142578, |
| "completions/min_length": 83.9, |
| "completions/min_terminated_length": 83.9, |
| "entropy": 0.1960208318196237, |
| "epoch": 0.47117516629711753, |
| "frac_reward_zero_std": 0.6375, |
| "grad_norm": 0.2001953125, |
| "learning_rate": 9.915100000000001e-06, |
| "loss": 0.0016, |
| "num_tokens": 80862634.0, |
| "reward": 0.30988283157348634, |
| "reward_std": 0.40548387467861174, |
| "rewards/reward_accuracy/mean": 0.21171875, |
| "rewards/reward_accuracy/std": 0.4044633388519287, |
| "rewards/reward_format/mean": 0.09816406443715095, |
| "rewards/reward_format/std": 0.010727366339415312, |
| "sampling/importance_sampling_ratio/max": 2.8331236839294434, |
| "sampling/importance_sampling_ratio/mean": 0.79440358877182, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6658172726631164, |
| "sampling/sampling_logp_difference/mean": 0.010510715562850237, |
| "step": 850, |
| "step_time": 22.92526392848231 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 1947.7, |
| "completions/max_terminated_length": 1628.4, |
| "completions/mean_length": 488.040625, |
| "completions/mean_terminated_length": 473.7301818847656, |
| "completions/min_length": 83.7, |
| "completions/min_terminated_length": 83.7, |
| "entropy": 0.19097628593444824, |
| "epoch": 0.47671840354767187, |
| "frac_reward_zero_std": 0.55, |
| "grad_norm": 0.1435546875, |
| "learning_rate": 9.9141e-06, |
| "loss": 0.0046, |
| "num_tokens": 81796942.0, |
| "reward": 0.3558203339576721, |
| "reward_std": 0.4364795982837677, |
| "rewards/reward_accuracy/mean": 0.2578125, |
| "rewards/reward_accuracy/std": 0.4352079004049301, |
| "rewards/reward_format/mean": 0.09800781458616256, |
| "rewards/reward_format/std": 0.010120696853846312, |
| "sampling/importance_sampling_ratio/max": 2.882861280441284, |
| "sampling/importance_sampling_ratio/mean": 0.8084916174411774, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.5413357198238373, |
| "sampling/sampling_logp_difference/mean": 0.010332701168954373, |
| "step": 860, |
| "step_time": 18.280066490639 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2589.7, |
| "completions/max_terminated_length": 2047.7, |
| "completions/mean_length": 528.32265625, |
| "completions/mean_terminated_length": 506.4889892578125, |
| "completions/min_length": 102.9, |
| "completions/min_terminated_length": 102.9, |
| "entropy": 0.19784362660720944, |
| "epoch": 0.48226164079822614, |
| "frac_reward_zero_std": 0.5375, |
| "grad_norm": 0.1953125, |
| "learning_rate": 9.913100000000002e-06, |
| "loss": -0.0019, |
| "num_tokens": 82776723.0, |
| "reward": 0.35253908336162565, |
| "reward_std": 0.42093836665153506, |
| "rewards/reward_accuracy/mean": 0.2546875, |
| "rewards/reward_accuracy/std": 0.4199315428733826, |
| "rewards/reward_format/mean": 0.09785156473517417, |
| "rewards/reward_format/std": 0.011529088206589221, |
| "sampling/importance_sampling_ratio/max": 2.7405781984329223, |
| "sampling/importance_sampling_ratio/mean": 0.781331044435501, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.8030954122543335, |
| "sampling/sampling_logp_difference/mean": 0.010619225073605775, |
| "step": 870, |
| "step_time": 24.139852070109917 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2785.1, |
| "completions/max_terminated_length": 2043.4, |
| "completions/mean_length": 498.73671875, |
| "completions/mean_terminated_length": 480.5369903564453, |
| "completions/min_length": 70.0, |
| "completions/min_terminated_length": 70.0, |
| "entropy": 0.1917380666360259, |
| "epoch": 0.4878048780487805, |
| "frac_reward_zero_std": 0.50625, |
| "grad_norm": 0.267578125, |
| "learning_rate": 9.912100000000001e-06, |
| "loss": 0.0122, |
| "num_tokens": 83717154.0, |
| "reward": 0.40437501668930054, |
| "reward_std": 0.45694526433944704, |
| "rewards/reward_accuracy/mean": 0.30625, |
| "rewards/reward_accuracy/std": 0.4560576915740967, |
| "rewards/reward_format/mean": 0.0981250025331974, |
| "rewards/reward_format/std": 0.010879392642527819, |
| "sampling/importance_sampling_ratio/max": 2.7803975105285645, |
| "sampling/importance_sampling_ratio/mean": 0.7973617076873779, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6012057065963745, |
| "sampling/sampling_logp_difference/mean": 0.01003720285370946, |
| "step": 880, |
| "step_time": 25.418268522666768 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2445.8, |
| "completions/max_terminated_length": 1915.2, |
| "completions/mean_length": 528.071875, |
| "completions/mean_terminated_length": 514.2183898925781, |
| "completions/min_length": 90.7, |
| "completions/min_terminated_length": 90.7, |
| "entropy": 0.18875117730349303, |
| "epoch": 0.4933481152993348, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.1923828125, |
| "learning_rate": 9.9111e-06, |
| "loss": -0.0088, |
| "num_tokens": 84703702.0, |
| "reward": 0.34289063811302184, |
| "reward_std": 0.4273763597011566, |
| "rewards/reward_accuracy/mean": 0.2453125, |
| "rewards/reward_accuracy/std": 0.4260083198547363, |
| "rewards/reward_format/mean": 0.09757812768220901, |
| "rewards/reward_format/std": 0.01137096043676138, |
| "sampling/importance_sampling_ratio/max": 2.723431444168091, |
| "sampling/importance_sampling_ratio/mean": 0.7746698379516601, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.697588312625885, |
| "sampling/sampling_logp_difference/mean": 0.010256011504679918, |
| "step": 890, |
| "step_time": 23.03028914211318 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2684.8, |
| "completions/max_terminated_length": 1858.0, |
| "completions/mean_length": 490.99296875, |
| "completions/mean_terminated_length": 470.7858123779297, |
| "completions/min_length": 94.8, |
| "completions/min_terminated_length": 94.8, |
| "entropy": 0.19142137542366983, |
| "epoch": 0.49889135254988914, |
| "frac_reward_zero_std": 0.60625, |
| "grad_norm": 0.1962890625, |
| "learning_rate": 9.9101e-06, |
| "loss": -0.0117, |
| "num_tokens": 85629205.0, |
| "reward": 0.3666015833616257, |
| "reward_std": 0.43854039907455444, |
| "rewards/reward_accuracy/mean": 0.26875, |
| "rewards/reward_accuracy/std": 0.43768568336963654, |
| "rewards/reward_format/mean": 0.09785156548023224, |
| "rewards/reward_format/std": 0.011392970057204365, |
| "sampling/importance_sampling_ratio/max": 2.696178603172302, |
| "sampling/importance_sampling_ratio/mean": 0.7830804288387299, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7175160884857178, |
| "sampling/sampling_logp_difference/mean": 0.010072064865380526, |
| "step": 900, |
| "step_time": 25.270411904109643 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2663.8, |
| "completions/max_terminated_length": 2038.2, |
| "completions/mean_length": 559.92734375, |
| "completions/mean_terminated_length": 544.1223937988282, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.19539868077263237, |
| "epoch": 0.5044345898004434, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.2275390625, |
| "learning_rate": 9.9091e-06, |
| "loss": -0.0001, |
| "num_tokens": 86660384.0, |
| "reward": 0.3096484556794167, |
| "reward_std": 0.3970925658941269, |
| "rewards/reward_accuracy/mean": 0.2109375, |
| "rewards/reward_accuracy/std": 0.39644255936145784, |
| "rewards/reward_format/mean": 0.0987109400331974, |
| "rewards/reward_format/std": 0.007910250313580036, |
| "sampling/importance_sampling_ratio/max": 2.5929040193557737, |
| "sampling/importance_sampling_ratio/mean": 0.7776078343391418, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7114342093467713, |
| "sampling/sampling_logp_difference/mean": 0.010492783412337304, |
| "step": 910, |
| "step_time": 25.07129647694528 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2909.8, |
| "completions/max_terminated_length": 1891.7, |
| "completions/mean_length": 526.19296875, |
| "completions/mean_terminated_length": 495.92922058105466, |
| "completions/min_length": 97.1, |
| "completions/min_terminated_length": 97.1, |
| "entropy": 0.18575275903567673, |
| "epoch": 0.5099778270509978, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.12890625, |
| "learning_rate": 9.908100000000001e-06, |
| "loss": -0.0005, |
| "num_tokens": 87632607.0, |
| "reward": 0.370976585149765, |
| "reward_std": 0.43615861535072326, |
| "rewards/reward_accuracy/mean": 0.2734375, |
| "rewards/reward_accuracy/std": 0.43485380709171295, |
| "rewards/reward_format/mean": 0.09753906428813934, |
| "rewards/reward_format/std": 0.013342957105487585, |
| "sampling/importance_sampling_ratio/max": 2.8099419355392454, |
| "sampling/importance_sampling_ratio/mean": 0.7895125448703766, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.610990047454834, |
| "sampling/sampling_logp_difference/mean": 0.009764214139431715, |
| "step": 920, |
| "step_time": 27.356456260895357 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2510.1, |
| "completions/max_terminated_length": 1868.6, |
| "completions/mean_length": 527.01484375, |
| "completions/mean_terminated_length": 513.1905090332032, |
| "completions/min_length": 98.4, |
| "completions/min_terminated_length": 98.4, |
| "entropy": 0.19027614817023278, |
| "epoch": 0.5155210643015521, |
| "frac_reward_zero_std": 0.65625, |
| "grad_norm": 0.1640625, |
| "learning_rate": 9.9071e-06, |
| "loss": 0.0135, |
| "num_tokens": 88620354.0, |
| "reward": 0.3265625208616257, |
| "reward_std": 0.4027924031019211, |
| "rewards/reward_accuracy/mean": 0.228125, |
| "rewards/reward_accuracy/std": 0.40238072276115416, |
| "rewards/reward_format/mean": 0.09843750074505805, |
| "rewards/reward_format/std": 0.009636348485946656, |
| "sampling/importance_sampling_ratio/max": 2.872739291191101, |
| "sampling/importance_sampling_ratio/mean": 0.7844687223434448, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7420347213745118, |
| "sampling/sampling_logp_difference/mean": 0.010081200953572988, |
| "step": 930, |
| "step_time": 24.056015314161776 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2583.4, |
| "completions/max_terminated_length": 2037.3, |
| "completions/mean_length": 535.23359375, |
| "completions/mean_terminated_length": 519.295297241211, |
| "completions/min_length": 107.2, |
| "completions/min_terminated_length": 107.2, |
| "entropy": 0.18940583611838518, |
| "epoch": 0.5210643015521065, |
| "frac_reward_zero_std": 0.65625, |
| "grad_norm": 0.1162109375, |
| "learning_rate": 9.9061e-06, |
| "loss": -0.0036, |
| "num_tokens": 89611877.0, |
| "reward": 0.333437517285347, |
| "reward_std": 0.4133425742387772, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.4124870508909225, |
| "rewards/reward_format/mean": 0.09828125089406967, |
| "rewards/reward_format/std": 0.010346282832324505, |
| "sampling/importance_sampling_ratio/max": 2.8139399766921995, |
| "sampling/importance_sampling_ratio/mean": 0.7930775284767151, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7239062309265136, |
| "sampling/sampling_logp_difference/mean": 0.010192825458943844, |
| "step": 940, |
| "step_time": 23.963245241809638 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2683.8, |
| "completions/max_terminated_length": 2137.3, |
| "completions/mean_length": 535.08046875, |
| "completions/mean_terminated_length": 517.3064544677734, |
| "completions/min_length": 78.0, |
| "completions/min_terminated_length": 78.0, |
| "entropy": 0.18976054461672903, |
| "epoch": 0.5266075388026608, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.12255859375, |
| "learning_rate": 9.905100000000001e-06, |
| "loss": -0.0075, |
| "num_tokens": 90599868.0, |
| "reward": 0.3147265791893005, |
| "reward_std": 0.4037634521722794, |
| "rewards/reward_accuracy/mean": 0.21640625, |
| "rewards/reward_accuracy/std": 0.4032208025455475, |
| "rewards/reward_format/mean": 0.09832031428813934, |
| "rewards/reward_format/std": 0.010422401316463947, |
| "sampling/importance_sampling_ratio/max": 2.7827829122543335, |
| "sampling/importance_sampling_ratio/mean": 0.7804066240787506, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7530510723590851, |
| "sampling/sampling_logp_difference/mean": 0.010086060408502818, |
| "step": 950, |
| "step_time": 25.02514584599994 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 2684.8, |
| "completions/max_terminated_length": 1903.1, |
| "completions/mean_length": 539.7609375, |
| "completions/mean_terminated_length": 500.4351776123047, |
| "completions/min_length": 102.8, |
| "completions/min_terminated_length": 102.8, |
| "entropy": 0.17585812751203775, |
| "epoch": 0.532150776053215, |
| "frac_reward_zero_std": 0.55, |
| "grad_norm": 0.1494140625, |
| "learning_rate": 9.9041e-06, |
| "loss": -0.0161, |
| "num_tokens": 91593754.0, |
| "reward": 0.3542968899011612, |
| "reward_std": 0.43039362132549286, |
| "rewards/reward_accuracy/mean": 0.25703125, |
| "rewards/reward_accuracy/std": 0.42896202206611633, |
| "rewards/reward_format/mean": 0.09726562574505807, |
| "rewards/reward_format/std": 0.013411387614905835, |
| "sampling/importance_sampling_ratio/max": 2.8197322368621824, |
| "sampling/importance_sampling_ratio/mean": 0.8465208590030671, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6621349811553955, |
| "sampling/sampling_logp_difference/mean": 0.009323875792324543, |
| "step": 960, |
| "step_time": 25.655348341632635 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01328125, |
| "completions/max_length": 2641.1, |
| "completions/max_terminated_length": 1900.6, |
| "completions/mean_length": 558.28359375, |
| "completions/mean_terminated_length": 526.1020751953125, |
| "completions/min_length": 89.9, |
| "completions/min_terminated_length": 89.9, |
| "entropy": 0.18624719949439167, |
| "epoch": 0.5376940133037694, |
| "frac_reward_zero_std": 0.525, |
| "grad_norm": 0.302734375, |
| "learning_rate": 9.903100000000002e-06, |
| "loss": -0.0026, |
| "num_tokens": 92624365.0, |
| "reward": 0.31886720806360247, |
| "reward_std": 0.41154763102531433, |
| "rewards/reward_accuracy/mean": 0.221875, |
| "rewards/reward_accuracy/std": 0.41036576628684995, |
| "rewards/reward_format/mean": 0.09699218869209289, |
| "rewards/reward_format/std": 0.014092974318191408, |
| "sampling/importance_sampling_ratio/max": 2.78151798248291, |
| "sampling/importance_sampling_ratio/mean": 0.7966257333755493, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.840751838684082, |
| "sampling/sampling_logp_difference/mean": 0.009627442061901092, |
| "step": 970, |
| "step_time": 25.72503827046603 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01328125, |
| "completions/max_length": 2653.2, |
| "completions/max_terminated_length": 2063.3, |
| "completions/mean_length": 531.353125, |
| "completions/mean_terminated_length": 497.41380004882814, |
| "completions/min_length": 102.5, |
| "completions/min_terminated_length": 102.5, |
| "entropy": 0.18513464787974954, |
| "epoch": 0.5432372505543237, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.130859375, |
| "learning_rate": 9.902100000000001e-06, |
| "loss": -0.0101, |
| "num_tokens": 93603593.0, |
| "reward": 0.35949220657348635, |
| "reward_std": 0.43016180396080017, |
| "rewards/reward_accuracy/mean": 0.26171875, |
| "rewards/reward_accuracy/std": 0.4288755238056183, |
| "rewards/reward_format/mean": 0.0977734386920929, |
| "rewards/reward_format/std": 0.011601851135492326, |
| "sampling/importance_sampling_ratio/max": 2.8363115072250364, |
| "sampling/importance_sampling_ratio/mean": 0.8088122189044953, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7168501287698745, |
| "sampling/sampling_logp_difference/mean": 0.009598701912909745, |
| "step": 980, |
| "step_time": 25.064363449905066 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2373.4, |
| "completions/max_terminated_length": 1992.7, |
| "completions/mean_length": 534.00234375, |
| "completions/mean_terminated_length": 515.8342498779297, |
| "completions/min_length": 102.9, |
| "completions/min_terminated_length": 102.9, |
| "entropy": 0.18777791429311036, |
| "epoch": 0.5487804878048781, |
| "frac_reward_zero_std": 0.5625, |
| "grad_norm": 0.1279296875, |
| "learning_rate": 9.9011e-06, |
| "loss": -0.0017, |
| "num_tokens": 94589468.0, |
| "reward": 0.3562109559774399, |
| "reward_std": 0.4181459844112396, |
| "rewards/reward_accuracy/mean": 0.2578125, |
| "rewards/reward_accuracy/std": 0.4171880155801773, |
| "rewards/reward_format/mean": 0.09839843884110451, |
| "rewards/reward_format/std": 0.009446130506694317, |
| "sampling/importance_sampling_ratio/max": 2.7549575090408327, |
| "sampling/importance_sampling_ratio/mean": 0.7960437357425689, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6936275601387024, |
| "sampling/sampling_logp_difference/mean": 0.010022966284304857, |
| "step": 990, |
| "step_time": 22.312294147955253 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2590.9, |
| "completions/max_terminated_length": 2230.5, |
| "completions/mean_length": 517.12265625, |
| "completions/mean_terminated_length": 505.02245178222654, |
| "completions/min_length": 99.5, |
| "completions/min_terminated_length": 99.5, |
| "entropy": 0.1894825303927064, |
| "epoch": 0.5543237250554324, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.357421875, |
| "learning_rate": 9.9001e-06, |
| "loss": -0.0184, |
| "num_tokens": 95548489.0, |
| "reward": 0.41816408932209015, |
| "reward_std": 0.4632305592298508, |
| "rewards/reward_accuracy/mean": 0.31953125, |
| "rewards/reward_accuracy/std": 0.46239044666290285, |
| "rewards/reward_format/mean": 0.09863281473517418, |
| "rewards/reward_format/std": 0.00944216875359416, |
| "sampling/importance_sampling_ratio/max": 2.720514512062073, |
| "sampling/importance_sampling_ratio/mean": 0.8006263434886932, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7036865711212158, |
| "sampling/sampling_logp_difference/mean": 0.01007543047890067, |
| "step": 1000, |
| "step_time": 24.32167164501734 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2751.9, |
| "completions/max_terminated_length": 2094.0, |
| "completions/mean_length": 554.26640625, |
| "completions/mean_terminated_length": 534.5031921386719, |
| "completions/min_length": 128.5, |
| "completions/min_terminated_length": 128.5, |
| "entropy": 0.18587267687544226, |
| "epoch": 0.5598669623059866, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.08447265625, |
| "learning_rate": 9.8991e-06, |
| "loss": -0.0162, |
| "num_tokens": 96566134.0, |
| "reward": 0.3615234553813934, |
| "reward_std": 0.43779799342155457, |
| "rewards/reward_accuracy/mean": 0.26328125, |
| "rewards/reward_accuracy/std": 0.4369611620903015, |
| "rewards/reward_format/mean": 0.09824218899011612, |
| "rewards/reward_format/std": 0.010238159354776144, |
| "sampling/importance_sampling_ratio/max": 2.790239691734314, |
| "sampling/importance_sampling_ratio/mean": 0.7736215829849243, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6948943912982941, |
| "sampling/sampling_logp_difference/mean": 0.00990040199831128, |
| "step": 1010, |
| "step_time": 25.536798378359528 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2218.0, |
| "completions/max_terminated_length": 2117.0, |
| "completions/mean_length": 543.7015625, |
| "completions/mean_terminated_length": 536.1788452148437, |
| "completions/min_length": 125.1, |
| "completions/min_terminated_length": 125.1, |
| "entropy": 0.1845726200379431, |
| "epoch": 0.565410199556541, |
| "frac_reward_zero_std": 0.60625, |
| "grad_norm": 0.162109375, |
| "learning_rate": 9.898100000000001e-06, |
| "loss": -0.023, |
| "num_tokens": 97569552.0, |
| "reward": 0.3337500184774399, |
| "reward_std": 0.4140059441328049, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.4132175028324127, |
| "rewards/reward_format/mean": 0.09859375208616257, |
| "rewards/reward_format/std": 0.009281962132081389, |
| "sampling/importance_sampling_ratio/max": 2.7193148374557494, |
| "sampling/importance_sampling_ratio/mean": 0.8077420055866241, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6622745990753174, |
| "sampling/sampling_logp_difference/mean": 0.009640573803335429, |
| "step": 1020, |
| "step_time": 20.971151125943287 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2752.7, |
| "completions/max_terminated_length": 1944.8, |
| "completions/mean_length": 581.98046875, |
| "completions/mean_terminated_length": 558.4423980712891, |
| "completions/min_length": 102.0, |
| "completions/min_terminated_length": 102.0, |
| "entropy": 0.17823353596031666, |
| "epoch": 0.5709534368070953, |
| "frac_reward_zero_std": 0.64375, |
| "grad_norm": 0.15234375, |
| "learning_rate": 9.8971e-06, |
| "loss": 0.0054, |
| "num_tokens": 98622071.0, |
| "reward": 0.3208984524011612, |
| "reward_std": 0.4060726135969162, |
| "rewards/reward_accuracy/mean": 0.22265625, |
| "rewards/reward_accuracy/std": 0.4053465068340302, |
| "rewards/reward_format/mean": 0.09824218899011612, |
| "rewards/reward_format/std": 0.010591733921319246, |
| "sampling/importance_sampling_ratio/max": 2.7654334545135497, |
| "sampling/importance_sampling_ratio/mean": 0.8025928676128388, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7992448151111603, |
| "sampling/sampling_logp_difference/mean": 0.0095288110896945, |
| "step": 1030, |
| "step_time": 26.087868539523335 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2390.2, |
| "completions/max_terminated_length": 1885.3, |
| "completions/mean_length": 519.33984375, |
| "completions/mean_terminated_length": 501.4436340332031, |
| "completions/min_length": 82.4, |
| "completions/min_terminated_length": 82.4, |
| "entropy": 0.1839101878926158, |
| "epoch": 0.5764966740576497, |
| "frac_reward_zero_std": 0.525, |
| "grad_norm": 0.59375, |
| "learning_rate": 9.8961e-06, |
| "loss": -0.0029, |
| "num_tokens": 99577682.0, |
| "reward": 0.3632031470537186, |
| "reward_std": 0.4346582591533661, |
| "rewards/reward_accuracy/mean": 0.265625, |
| "rewards/reward_accuracy/std": 0.4339329838752747, |
| "rewards/reward_format/mean": 0.09757812693715096, |
| "rewards/reward_format/std": 0.012111793830990791, |
| "sampling/importance_sampling_ratio/max": 2.71311411857605, |
| "sampling/importance_sampling_ratio/mean": 0.8175545930862427, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7128188014030457, |
| "sampling/sampling_logp_difference/mean": 0.009574778471142053, |
| "step": 1040, |
| "step_time": 22.255110158678143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2578.9, |
| "completions/max_terminated_length": 1962.9, |
| "completions/mean_length": 537.440625, |
| "completions/mean_terminated_length": 517.59228515625, |
| "completions/min_length": 114.5, |
| "completions/min_terminated_length": 114.5, |
| "entropy": 0.1768434618599713, |
| "epoch": 0.582039911308204, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.2890625, |
| "learning_rate": 9.895100000000001e-06, |
| "loss": -0.01, |
| "num_tokens": 100569030.0, |
| "reward": 0.3810937628149986, |
| "reward_std": 0.43545140624046325, |
| "rewards/reward_accuracy/mean": 0.2828125, |
| "rewards/reward_accuracy/std": 0.4346814572811127, |
| "rewards/reward_format/mean": 0.09828125014901161, |
| "rewards/reward_format/std": 0.010867481213063001, |
| "sampling/importance_sampling_ratio/max": 2.810622453689575, |
| "sampling/importance_sampling_ratio/mean": 0.7989472687244416, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6357433915138244, |
| "sampling/sampling_logp_difference/mean": 0.009369999915361405, |
| "step": 1050, |
| "step_time": 23.854101517796515 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2588.9, |
| "completions/max_terminated_length": 1970.8, |
| "completions/mean_length": 562.16171875, |
| "completions/mean_terminated_length": 534.3940551757812, |
| "completions/min_length": 117.0, |
| "completions/min_terminated_length": 117.0, |
| "entropy": 0.18549765599891543, |
| "epoch": 0.5875831485587583, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.2314453125, |
| "learning_rate": 9.8941e-06, |
| "loss": -0.0061, |
| "num_tokens": 101597261.0, |
| "reward": 0.3712500140070915, |
| "reward_std": 0.4297492831945419, |
| "rewards/reward_accuracy/mean": 0.27421875, |
| "rewards/reward_accuracy/std": 0.4284309297800064, |
| "rewards/reward_format/mean": 0.09703125208616256, |
| "rewards/reward_format/std": 0.013438827963545919, |
| "sampling/importance_sampling_ratio/max": 2.723525953292847, |
| "sampling/importance_sampling_ratio/mean": 0.7830065906047821, |
| "sampling/importance_sampling_ratio/min": 0.013571934401988983, |
| "sampling/sampling_logp_difference/max": 0.7980721831321717, |
| "sampling/sampling_logp_difference/mean": 0.009784030448645353, |
| "step": 1060, |
| "step_time": 25.33591081867926 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2644.4, |
| "completions/max_terminated_length": 2065.4, |
| "completions/mean_length": 517.43359375, |
| "completions/mean_terminated_length": 503.3719757080078, |
| "completions/min_length": 109.3, |
| "completions/min_terminated_length": 109.3, |
| "entropy": 0.18392504313960673, |
| "epoch": 0.5931263858093127, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.1240234375, |
| "learning_rate": 9.8931e-06, |
| "loss": -0.0129, |
| "num_tokens": 102560440.0, |
| "reward": 0.3447656393051147, |
| "reward_std": 0.423628443479538, |
| "rewards/reward_accuracy/mean": 0.24609375, |
| "rewards/reward_accuracy/std": 0.4229851454496384, |
| "rewards/reward_format/mean": 0.09867187514901161, |
| "rewards/reward_format/std": 0.009405081253498792, |
| "sampling/importance_sampling_ratio/max": 2.721461033821106, |
| "sampling/importance_sampling_ratio/mean": 0.80303915143013, |
| "sampling/importance_sampling_ratio/min": 0.0010761947371065617, |
| "sampling/sampling_logp_difference/max": 0.6841929852962494, |
| "sampling/sampling_logp_difference/mean": 0.00962362289428711, |
| "step": 1070, |
| "step_time": 24.441998413717375 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2825.2, |
| "completions/max_terminated_length": 1999.6, |
| "completions/mean_length": 530.22265625, |
| "completions/mean_terminated_length": 508.48065795898435, |
| "completions/min_length": 92.2, |
| "completions/min_terminated_length": 92.2, |
| "entropy": 0.1856986002996564, |
| "epoch": 0.5986696230598669, |
| "frac_reward_zero_std": 0.60625, |
| "grad_norm": 0.1796875, |
| "learning_rate": 9.892100000000001e-06, |
| "loss": -0.0078, |
| "num_tokens": 103540909.0, |
| "reward": 0.32304688841104506, |
| "reward_std": 0.4056141063570976, |
| "rewards/reward_accuracy/mean": 0.225, |
| "rewards/reward_accuracy/std": 0.4049458190798759, |
| "rewards/reward_format/mean": 0.09804687723517418, |
| "rewards/reward_format/std": 0.010351290460675955, |
| "sampling/importance_sampling_ratio/max": 2.6503360509872436, |
| "sampling/importance_sampling_ratio/mean": 0.8159402847290039, |
| "sampling/importance_sampling_ratio/min": 0.002789578586816788, |
| "sampling/sampling_logp_difference/max": 0.7264703810214996, |
| "sampling/sampling_logp_difference/mean": 0.009821361117064952, |
| "step": 1080, |
| "step_time": 25.980040215374903 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2518.2, |
| "completions/max_terminated_length": 2194.1, |
| "completions/mean_length": 532.33515625, |
| "completions/mean_terminated_length": 520.2329040527344, |
| "completions/min_length": 91.8, |
| "completions/min_terminated_length": 91.8, |
| "entropy": 0.19024350475519897, |
| "epoch": 0.6042128603104213, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.134765625, |
| "learning_rate": 9.891100000000001e-06, |
| "loss": 0.0033, |
| "num_tokens": 104528706.0, |
| "reward": 0.32835939824581145, |
| "reward_std": 0.40663976073265073, |
| "rewards/reward_accuracy/mean": 0.2296875, |
| "rewards/reward_accuracy/std": 0.40600984543561935, |
| "rewards/reward_format/mean": 0.09867187738418579, |
| "rewards/reward_format/std": 0.00926654487848282, |
| "sampling/importance_sampling_ratio/max": 2.822052240371704, |
| "sampling/importance_sampling_ratio/mean": 0.791909146308899, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6315569877624512, |
| "sampling/sampling_logp_difference/mean": 0.010114913620054723, |
| "step": 1090, |
| "step_time": 23.578632047353313 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2746.1, |
| "completions/max_terminated_length": 1999.2, |
| "completions/mean_length": 518.7890625, |
| "completions/mean_terminated_length": 500.6765380859375, |
| "completions/min_length": 90.8, |
| "completions/min_terminated_length": 90.8, |
| "entropy": 0.18212580690160393, |
| "epoch": 0.6097560975609756, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.1689453125, |
| "learning_rate": 9.8901e-06, |
| "loss": -0.0081, |
| "num_tokens": 105497676.0, |
| "reward": 0.3657812625169754, |
| "reward_std": 0.4306503027677536, |
| "rewards/reward_accuracy/mean": 0.26796875, |
| "rewards/reward_accuracy/std": 0.4295253038406372, |
| "rewards/reward_format/mean": 0.09781250208616257, |
| "rewards/reward_format/std": 0.011111386446282267, |
| "sampling/importance_sampling_ratio/max": 2.7908746004104614, |
| "sampling/importance_sampling_ratio/mean": 0.8427472233772277, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6617484807968139, |
| "sampling/sampling_logp_difference/mean": 0.009653772879391908, |
| "step": 1100, |
| "step_time": 25.698698367876933 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2679.4, |
| "completions/max_terminated_length": 2024.5, |
| "completions/mean_length": 518.415625, |
| "completions/mean_terminated_length": 488.26637878417966, |
| "completions/min_length": 100.6, |
| "completions/min_terminated_length": 100.6, |
| "entropy": 0.18338730297982692, |
| "epoch": 0.61529933481153, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.10107421875, |
| "learning_rate": 9.8891e-06, |
| "loss": -0.0094, |
| "num_tokens": 106465216.0, |
| "reward": 0.3871093988418579, |
| "reward_std": 0.44868123829364776, |
| "rewards/reward_accuracy/mean": 0.2890625, |
| "rewards/reward_accuracy/std": 0.4477301865816116, |
| "rewards/reward_format/mean": 0.09804687574505806, |
| "rewards/reward_format/std": 0.011667386116459965, |
| "sampling/importance_sampling_ratio/max": 2.814854884147644, |
| "sampling/importance_sampling_ratio/mean": 0.8119877338409424, |
| "sampling/importance_sampling_ratio/min": 0.004733159765601158, |
| "sampling/sampling_logp_difference/max": 0.6506388306617736, |
| "sampling/sampling_logp_difference/mean": 0.009583959821611643, |
| "step": 1110, |
| "step_time": 25.045554387150332 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2732.1, |
| "completions/max_terminated_length": 2191.4, |
| "completions/mean_length": 554.70625, |
| "completions/mean_terminated_length": 531.1011840820313, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.17923241034150122, |
| "epoch": 0.6208425720620843, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.1875, |
| "learning_rate": 9.888100000000001e-06, |
| "loss": 0.0021, |
| "num_tokens": 107480448.0, |
| "reward": 0.33285157531499865, |
| "reward_std": 0.4102482095360756, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.4090228483080864, |
| "rewards/reward_format/mean": 0.09769531413912773, |
| "rewards/reward_format/std": 0.0118531777523458, |
| "sampling/importance_sampling_ratio/max": 2.71269805431366, |
| "sampling/importance_sampling_ratio/mean": 0.8175427377223968, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6922570943832398, |
| "sampling/sampling_logp_difference/mean": 0.009599831979721784, |
| "step": 1120, |
| "step_time": 25.72594743198715 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01640625, |
| "completions/max_length": 2901.4, |
| "completions/max_terminated_length": 2218.8, |
| "completions/mean_length": 569.0359375, |
| "completions/mean_terminated_length": 527.0262786865235, |
| "completions/min_length": 118.0, |
| "completions/min_terminated_length": 118.0, |
| "entropy": 0.17688990677706898, |
| "epoch": 0.6263858093126385, |
| "frac_reward_zero_std": 0.475, |
| "grad_norm": 0.255859375, |
| "learning_rate": 9.8871e-06, |
| "loss": 0.0203, |
| "num_tokens": 108506150.0, |
| "reward": 0.3756640762090683, |
| "reward_std": 0.4380262762308121, |
| "rewards/reward_accuracy/mean": 0.27890625, |
| "rewards/reward_accuracy/std": 0.43619746565818784, |
| "rewards/reward_format/mean": 0.09675781354308129, |
| "rewards/reward_format/std": 0.015205803513526916, |
| "sampling/importance_sampling_ratio/max": 2.7113341093063354, |
| "sampling/importance_sampling_ratio/mean": 0.7919621288776397, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6687087774276733, |
| "sampling/sampling_logp_difference/mean": 0.009188767662271858, |
| "step": 1130, |
| "step_time": 27.586753958277406 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2672.9, |
| "completions/max_terminated_length": 1698.8, |
| "completions/mean_length": 541.4765625, |
| "completions/mean_terminated_length": 515.5257141113282, |
| "completions/min_length": 101.8, |
| "completions/min_terminated_length": 101.8, |
| "entropy": 0.17366706570610405, |
| "epoch": 0.6319290465631929, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.1416015625, |
| "learning_rate": 9.8861e-06, |
| "loss": -0.0074, |
| "num_tokens": 109496976.0, |
| "reward": 0.36753908097743987, |
| "reward_std": 0.4417306512594223, |
| "rewards/reward_accuracy/mean": 0.26953125, |
| "rewards/reward_accuracy/std": 0.4403674483299255, |
| "rewards/reward_format/mean": 0.09800781533122063, |
| "rewards/reward_format/std": 0.012234439281746745, |
| "sampling/importance_sampling_ratio/max": 2.7885345458984374, |
| "sampling/importance_sampling_ratio/mean": 0.8062734842300415, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.689586853981018, |
| "sampling/sampling_logp_difference/mean": 0.00918955635279417, |
| "step": 1140, |
| "step_time": 25.135763539513572 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2604.6, |
| "completions/max_terminated_length": 2097.3, |
| "completions/mean_length": 566.68046875, |
| "completions/mean_terminated_length": 548.841683959961, |
| "completions/min_length": 106.3, |
| "completions/min_terminated_length": 106.3, |
| "entropy": 0.17535352939739823, |
| "epoch": 0.6374722838137472, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.251953125, |
| "learning_rate": 9.885100000000001e-06, |
| "loss": -0.0046, |
| "num_tokens": 110529871.0, |
| "reward": 0.3267968952655792, |
| "reward_std": 0.4079524129629135, |
| "rewards/reward_accuracy/mean": 0.228125, |
| "rewards/reward_accuracy/std": 0.4071876615285873, |
| "rewards/reward_format/mean": 0.09867187738418579, |
| "rewards/reward_format/std": 0.008174518495798111, |
| "sampling/importance_sampling_ratio/max": 2.787992262840271, |
| "sampling/importance_sampling_ratio/mean": 0.8012015223503113, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7749575853347779, |
| "sampling/sampling_logp_difference/mean": 0.009686916880309582, |
| "step": 1150, |
| "step_time": 24.567052013333885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2872.0, |
| "completions/max_terminated_length": 2115.5, |
| "completions/mean_length": 550.65390625, |
| "completions/mean_terminated_length": 528.7150970458985, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.1830344174988568, |
| "epoch": 0.6430155210643016, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.1748046875, |
| "learning_rate": 9.8841e-06, |
| "loss": -0.0003, |
| "num_tokens": 111546020.0, |
| "reward": 0.3435937613248825, |
| "reward_std": 0.4232655853033066, |
| "rewards/reward_accuracy/mean": 0.2453125, |
| "rewards/reward_accuracy/std": 0.4221104919910431, |
| "rewards/reward_format/mean": 0.09828125089406967, |
| "rewards/reward_format/std": 0.010212259879335762, |
| "sampling/importance_sampling_ratio/max": 2.7360779523849486, |
| "sampling/importance_sampling_ratio/mean": 0.8037152945995331, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6048502266407013, |
| "sampling/sampling_logp_difference/mean": 0.009786820970475674, |
| "step": 1160, |
| "step_time": 27.11026292219758 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2849.4, |
| "completions/max_terminated_length": 2102.9, |
| "completions/mean_length": 508.1953125, |
| "completions/mean_terminated_length": 490.06964111328125, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.173277937900275, |
| "epoch": 0.6485587583148559, |
| "frac_reward_zero_std": 0.55, |
| "grad_norm": 0.294921875, |
| "learning_rate": 9.8831e-06, |
| "loss": 0.0019, |
| "num_tokens": 112494534.0, |
| "reward": 0.4316015899181366, |
| "reward_std": 0.4690641850233078, |
| "rewards/reward_accuracy/mean": 0.3328125, |
| "rewards/reward_accuracy/std": 0.4681966692209244, |
| "rewards/reward_format/mean": 0.09878906533122063, |
| "rewards/reward_format/std": 0.009305673372000455, |
| "sampling/importance_sampling_ratio/max": 2.7979384422302247, |
| "sampling/importance_sampling_ratio/mean": 0.8239940285682679, |
| "sampling/importance_sampling_ratio/min": 0.004945734143257141, |
| "sampling/sampling_logp_difference/max": 0.7263288736343384, |
| "sampling/sampling_logp_difference/mean": 0.009251552633941174, |
| "step": 1170, |
| "step_time": 26.334344477858394 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2500.8, |
| "completions/max_terminated_length": 2096.9, |
| "completions/mean_length": 551.96875, |
| "completions/mean_terminated_length": 532.2167022705078, |
| "completions/min_length": 88.8, |
| "completions/min_terminated_length": 88.8, |
| "entropy": 0.18950509205460547, |
| "epoch": 0.6541019955654102, |
| "frac_reward_zero_std": 0.5625, |
| "grad_norm": 0.26953125, |
| "learning_rate": 9.882100000000001e-06, |
| "loss": -0.0112, |
| "num_tokens": 113512118.0, |
| "reward": 0.31457033157348635, |
| "reward_std": 0.40379429161548613, |
| "rewards/reward_accuracy/mean": 0.21640625, |
| "rewards/reward_accuracy/std": 0.40275146067142487, |
| "rewards/reward_format/mean": 0.09816406369209289, |
| "rewards/reward_format/std": 0.009288741322234274, |
| "sampling/importance_sampling_ratio/max": 2.8494544982910157, |
| "sampling/importance_sampling_ratio/mean": 0.7939520180225372, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7003115177154541, |
| "sampling/sampling_logp_difference/mean": 0.009875538293272257, |
| "step": 1180, |
| "step_time": 23.63708021636121 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2602.4, |
| "completions/max_terminated_length": 1929.6, |
| "completions/mean_length": 528.36328125, |
| "completions/mean_terminated_length": 508.5124481201172, |
| "completions/min_length": 113.9, |
| "completions/min_terminated_length": 113.9, |
| "entropy": 0.17443360132165253, |
| "epoch": 0.6596452328159645, |
| "frac_reward_zero_std": 0.6125, |
| "grad_norm": 0.224609375, |
| "learning_rate": 9.881100000000001e-06, |
| "loss": 0.0065, |
| "num_tokens": 114490503.0, |
| "reward": 0.3731250137090683, |
| "reward_std": 0.437736040353775, |
| "rewards/reward_accuracy/mean": 0.275, |
| "rewards/reward_accuracy/std": 0.4368744075298309, |
| "rewards/reward_format/mean": 0.09812500104308128, |
| "rewards/reward_format/std": 0.010072857653722168, |
| "sampling/importance_sampling_ratio/max": 2.8203094005584717, |
| "sampling/importance_sampling_ratio/mean": 0.8166678011417389, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7252306759357452, |
| "sampling/sampling_logp_difference/mean": 0.009223622642457486, |
| "step": 1190, |
| "step_time": 24.316923525743185 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2739.5, |
| "completions/max_terminated_length": 1929.5, |
| "completions/mean_length": 496.73046875, |
| "completions/mean_terminated_length": 466.22693786621096, |
| "completions/min_length": 97.1, |
| "completions/min_terminated_length": 97.1, |
| "entropy": 0.19221581825986506, |
| "epoch": 0.6651884700665188, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.2060546875, |
| "learning_rate": 9.880100000000002e-06, |
| "loss": 0.0017, |
| "num_tokens": 115436862.0, |
| "reward": 0.39460939913988113, |
| "reward_std": 0.4369867041707039, |
| "rewards/reward_accuracy/mean": 0.29609375, |
| "rewards/reward_accuracy/std": 0.43613731414079665, |
| "rewards/reward_format/mean": 0.09851562678813934, |
| "rewards/reward_format/std": 0.010253236815333366, |
| "sampling/importance_sampling_ratio/max": 2.758878231048584, |
| "sampling/importance_sampling_ratio/mean": 0.8236484527587891, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6555019378662109, |
| "sampling/sampling_logp_difference/mean": 0.009928112756460905, |
| "step": 1200, |
| "step_time": 25.75226068943739 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0125, |
| "completions/max_length": 2425.8, |
| "completions/max_terminated_length": 2210.3, |
| "completions/mean_length": 555.26875, |
| "completions/mean_terminated_length": 524.0920532226562, |
| "completions/min_length": 122.9, |
| "completions/min_terminated_length": 122.9, |
| "entropy": 0.16942479265853763, |
| "epoch": 0.6707317073170732, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.1376953125, |
| "learning_rate": 9.8791e-06, |
| "loss": -0.0088, |
| "num_tokens": 116448166.0, |
| "reward": 0.38503908962011335, |
| "reward_std": 0.43941148519515993, |
| "rewards/reward_accuracy/mean": 0.28671875, |
| "rewards/reward_accuracy/std": 0.4385279446840286, |
| "rewards/reward_format/mean": 0.09832031577825547, |
| "rewards/reward_format/std": 0.00951326610520482, |
| "sampling/importance_sampling_ratio/max": 2.858626937866211, |
| "sampling/importance_sampling_ratio/mean": 0.8329061985015869, |
| "sampling/importance_sampling_ratio/min": 0.0012984445318579673, |
| "sampling/sampling_logp_difference/max": 0.6829913079738616, |
| "sampling/sampling_logp_difference/mean": 0.00901740463450551, |
| "step": 1210, |
| "step_time": 23.930665357084944 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2727.6, |
| "completions/max_terminated_length": 2114.8, |
| "completions/mean_length": 548.084375, |
| "completions/mean_terminated_length": 526.177880859375, |
| "completions/min_length": 122.3, |
| "completions/min_terminated_length": 122.3, |
| "entropy": 0.18081455240026117, |
| "epoch": 0.6762749445676275, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.35546875, |
| "learning_rate": 9.878100000000001e-06, |
| "loss": -0.0133, |
| "num_tokens": 117462858.0, |
| "reward": 0.32019533812999723, |
| "reward_std": 0.39060440063476565, |
| "rewards/reward_accuracy/mean": 0.221875, |
| "rewards/reward_accuracy/std": 0.3895674705505371, |
| "rewards/reward_format/mean": 0.09832031652331352, |
| "rewards/reward_format/std": 0.01072123358026147, |
| "sampling/importance_sampling_ratio/max": 2.7042575120925902, |
| "sampling/importance_sampling_ratio/mean": 0.820432984828949, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7098974585533142, |
| "sampling/sampling_logp_difference/mean": 0.009510817192494869, |
| "step": 1220, |
| "step_time": 25.745407524192707 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2794.0, |
| "completions/max_terminated_length": 1786.4, |
| "completions/mean_length": 525.5171875, |
| "completions/mean_terminated_length": 501.4542999267578, |
| "completions/min_length": 112.7, |
| "completions/min_terminated_length": 112.7, |
| "entropy": 0.17705879975110292, |
| "epoch": 0.6818181818181818, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.1357421875, |
| "learning_rate": 9.8771e-06, |
| "loss": -0.0131, |
| "num_tokens": 118434872.0, |
| "reward": 0.3681640729308128, |
| "reward_std": 0.4274998664855957, |
| "rewards/reward_accuracy/mean": 0.26953125, |
| "rewards/reward_accuracy/std": 0.426558056473732, |
| "rewards/reward_format/mean": 0.09863281399011611, |
| "rewards/reward_format/std": 0.0097567493095994, |
| "sampling/importance_sampling_ratio/max": 2.876078414916992, |
| "sampling/importance_sampling_ratio/mean": 0.8416208446025848, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7303533673286438, |
| "sampling/sampling_logp_difference/mean": 0.009437979385256767, |
| "step": 1230, |
| "step_time": 26.65006161974743 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2869.0, |
| "completions/max_terminated_length": 2227.9, |
| "completions/mean_length": 546.8078125, |
| "completions/mean_terminated_length": 519.1445831298828, |
| "completions/min_length": 98.5, |
| "completions/min_terminated_length": 98.5, |
| "entropy": 0.18690253337845206, |
| "epoch": 0.6873614190687362, |
| "frac_reward_zero_std": 0.6125, |
| "grad_norm": 0.166015625, |
| "learning_rate": 9.8761e-06, |
| "loss": -0.0132, |
| "num_tokens": 119436378.0, |
| "reward": 0.37964846193790436, |
| "reward_std": 0.4379155933856964, |
| "rewards/reward_accuracy/mean": 0.28125, |
| "rewards/reward_accuracy/std": 0.43678098917007446, |
| "rewards/reward_format/mean": 0.09839843958616257, |
| "rewards/reward_format/std": 0.010618381900712848, |
| "sampling/importance_sampling_ratio/max": 2.717579650878906, |
| "sampling/importance_sampling_ratio/mean": 0.7808829426765442, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6815200865268707, |
| "sampling/sampling_logp_difference/mean": 0.009923304803669453, |
| "step": 1240, |
| "step_time": 27.079504994302987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2623.2, |
| "completions/max_terminated_length": 2126.5, |
| "completions/mean_length": 548.428125, |
| "completions/mean_terminated_length": 522.7127105712891, |
| "completions/min_length": 101.8, |
| "completions/min_terminated_length": 101.8, |
| "entropy": 0.1743779300712049, |
| "epoch": 0.6929046563192904, |
| "frac_reward_zero_std": 0.56875, |
| "grad_norm": 0.26953125, |
| "learning_rate": 9.875100000000001e-06, |
| "loss": 0.01, |
| "num_tokens": 120446446.0, |
| "reward": 0.3725781410932541, |
| "reward_std": 0.4461402207612991, |
| "rewards/reward_accuracy/mean": 0.27421875, |
| "rewards/reward_accuracy/std": 0.44498314559459684, |
| "rewards/reward_format/mean": 0.09835937693715095, |
| "rewards/reward_format/std": 0.009066996164619923, |
| "sampling/importance_sampling_ratio/max": 2.7299872636795044, |
| "sampling/importance_sampling_ratio/mean": 0.8279211938381195, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.9901257157325745, |
| "sampling/sampling_logp_difference/mean": 0.009429467935115099, |
| "step": 1250, |
| "step_time": 24.69690683884546 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2761.7, |
| "completions/max_terminated_length": 2202.8, |
| "completions/mean_length": 570.496875, |
| "completions/mean_terminated_length": 543.407080078125, |
| "completions/min_length": 129.9, |
| "completions/min_terminated_length": 129.9, |
| "entropy": 0.168515548016876, |
| "epoch": 0.6984478935698448, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.255859375, |
| "learning_rate": 9.874100000000001e-06, |
| "loss": -0.0116, |
| "num_tokens": 121493010.0, |
| "reward": 0.31601563692092893, |
| "reward_std": 0.4141286164522171, |
| "rewards/reward_accuracy/mean": 0.21796875, |
| "rewards/reward_accuracy/std": 0.4128956526517868, |
| "rewards/reward_format/mean": 0.098046875, |
| "rewards/reward_format/std": 0.010810937732458115, |
| "sampling/importance_sampling_ratio/max": 2.763057065010071, |
| "sampling/importance_sampling_ratio/mean": 0.793667197227478, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7122621536254883, |
| "sampling/sampling_logp_difference/mean": 0.009065543115139008, |
| "step": 1260, |
| "step_time": 26.383650957606733 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2527.6, |
| "completions/max_terminated_length": 2107.2, |
| "completions/mean_length": 510.15234375, |
| "completions/mean_terminated_length": 487.8351593017578, |
| "completions/min_length": 93.6, |
| "completions/min_terminated_length": 93.6, |
| "entropy": 0.18368760915473104, |
| "epoch": 0.7039911308203991, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.1748046875, |
| "learning_rate": 9.8731e-06, |
| "loss": -0.0042, |
| "num_tokens": 122449141.0, |
| "reward": 0.4032031446695328, |
| "reward_std": 0.45588718056678773, |
| "rewards/reward_accuracy/mean": 0.3046875, |
| "rewards/reward_accuracy/std": 0.45476851165294646, |
| "rewards/reward_format/mean": 0.09851562827825547, |
| "rewards/reward_format/std": 0.009620985947549343, |
| "sampling/importance_sampling_ratio/max": 2.833859753608704, |
| "sampling/importance_sampling_ratio/mean": 0.8182863891124725, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7316115021705627, |
| "sampling/sampling_logp_difference/mean": 0.009826147649437188, |
| "step": 1270, |
| "step_time": 23.82557516680099 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2754.4, |
| "completions/max_terminated_length": 2132.0, |
| "completions/mean_length": 544.12578125, |
| "completions/mean_terminated_length": 530.0158233642578, |
| "completions/min_length": 91.9, |
| "completions/min_terminated_length": 91.9, |
| "entropy": 0.18715481441468002, |
| "epoch": 0.7095343680709535, |
| "frac_reward_zero_std": 0.56875, |
| "grad_norm": 0.1572265625, |
| "learning_rate": 9.872100000000002e-06, |
| "loss": -0.0121, |
| "num_tokens": 123448646.0, |
| "reward": 0.361406272649765, |
| "reward_std": 0.4257373481988907, |
| "rewards/reward_accuracy/mean": 0.2625, |
| "rewards/reward_accuracy/std": 0.4252686381340027, |
| "rewards/reward_format/mean": 0.0989062525331974, |
| "rewards/reward_format/std": 0.007946221996098757, |
| "sampling/importance_sampling_ratio/max": 2.834231066703796, |
| "sampling/importance_sampling_ratio/mean": 0.8099711120128632, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7076910018920899, |
| "sampling/sampling_logp_difference/mean": 0.009653880633413792, |
| "step": 1280, |
| "step_time": 25.815507070487364 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2620.3, |
| "completions/max_terminated_length": 2021.6, |
| "completions/mean_length": 533.6640625, |
| "completions/mean_terminated_length": 511.88072509765624, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.17582522793672978, |
| "epoch": 0.7150776053215078, |
| "frac_reward_zero_std": 0.6375, |
| "grad_norm": 0.2578125, |
| "learning_rate": 9.871100000000001e-06, |
| "loss": 0.0032, |
| "num_tokens": 124436464.0, |
| "reward": 0.3636328294873238, |
| "reward_std": 0.43050728738307953, |
| "rewards/reward_accuracy/mean": 0.26484375, |
| "rewards/reward_accuracy/std": 0.42997671067714693, |
| "rewards/reward_format/mean": 0.0987890638411045, |
| "rewards/reward_format/std": 0.009447445347905158, |
| "sampling/importance_sampling_ratio/max": 2.7626739025115965, |
| "sampling/importance_sampling_ratio/mean": 0.8272136390209198, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6840312719345093, |
| "sampling/sampling_logp_difference/mean": 0.00935955811291933, |
| "step": 1290, |
| "step_time": 24.619464975828304 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2717.3, |
| "completions/max_terminated_length": 2081.4, |
| "completions/mean_length": 541.33359375, |
| "completions/mean_terminated_length": 517.4955078125, |
| "completions/min_length": 91.7, |
| "completions/min_terminated_length": 91.7, |
| "entropy": 0.174969047261402, |
| "epoch": 0.720620842572062, |
| "frac_reward_zero_std": 0.625, |
| "grad_norm": 0.1435546875, |
| "learning_rate": 9.8701e-06, |
| "loss": -0.0081, |
| "num_tokens": 125437779.0, |
| "reward": 0.36046876907348635, |
| "reward_std": 0.4397426903247833, |
| "rewards/reward_accuracy/mean": 0.26171875, |
| "rewards/reward_accuracy/std": 0.4388545870780945, |
| "rewards/reward_format/mean": 0.09875000268220901, |
| "rewards/reward_format/std": 0.009604779724031686, |
| "sampling/importance_sampling_ratio/max": 2.7446279287338258, |
| "sampling/importance_sampling_ratio/mean": 0.8435998082160949, |
| "sampling/importance_sampling_ratio/min": 0.002353167161345482, |
| "sampling/sampling_logp_difference/max": 0.7189603805541992, |
| "sampling/sampling_logp_difference/mean": 0.009142240416258574, |
| "step": 1300, |
| "step_time": 25.59506274233572 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2620.3, |
| "completions/max_terminated_length": 2022.8, |
| "completions/mean_length": 546.0390625, |
| "completions/mean_terminated_length": 522.5009552001953, |
| "completions/min_length": 87.6, |
| "completions/min_terminated_length": 87.6, |
| "entropy": 0.1756468765437603, |
| "epoch": 0.7261640798226164, |
| "frac_reward_zero_std": 0.675, |
| "grad_norm": 0.1796875, |
| "learning_rate": 9.8691e-06, |
| "loss": -0.0144, |
| "num_tokens": 126440317.0, |
| "reward": 0.34820314645767214, |
| "reward_std": 0.4247490048408508, |
| "rewards/reward_accuracy/mean": 0.25, |
| "rewards/reward_accuracy/std": 0.4237367570400238, |
| "rewards/reward_format/mean": 0.09820312708616256, |
| "rewards/reward_format/std": 0.010393007192760706, |
| "sampling/importance_sampling_ratio/max": 2.909200596809387, |
| "sampling/importance_sampling_ratio/mean": 0.8156356334686279, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7257403492927551, |
| "sampling/sampling_logp_difference/mean": 0.009307279624044896, |
| "step": 1310, |
| "step_time": 24.645184479188174 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2443.6, |
| "completions/max_terminated_length": 1843.6, |
| "completions/mean_length": 510.71640625, |
| "completions/mean_terminated_length": 498.8135070800781, |
| "completions/min_length": 99.7, |
| "completions/min_terminated_length": 99.7, |
| "entropy": 0.18384518059901894, |
| "epoch": 0.7317073170731707, |
| "frac_reward_zero_std": 0.66875, |
| "grad_norm": 0.13671875, |
| "learning_rate": 9.868100000000001e-06, |
| "loss": -0.0034, |
| "num_tokens": 127397226.0, |
| "reward": 0.3685156524181366, |
| "reward_std": 0.4367151528596878, |
| "rewards/reward_accuracy/mean": 0.26953125, |
| "rewards/reward_accuracy/std": 0.43615304827690127, |
| "rewards/reward_format/mean": 0.09898437932133675, |
| "rewards/reward_format/std": 0.007863423228263855, |
| "sampling/importance_sampling_ratio/max": 2.857421112060547, |
| "sampling/importance_sampling_ratio/mean": 0.8453933894634247, |
| "sampling/importance_sampling_ratio/min": 0.004858009889721871, |
| "sampling/sampling_logp_difference/max": 0.6443193912506103, |
| "sampling/sampling_logp_difference/mean": 0.009931729454547168, |
| "step": 1320, |
| "step_time": 22.742084157746284 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2691.8, |
| "completions/max_terminated_length": 2164.2, |
| "completions/mean_length": 534.93125, |
| "completions/mean_terminated_length": 508.84022827148436, |
| "completions/min_length": 82.7, |
| "completions/min_terminated_length": 82.7, |
| "entropy": 0.18140620556659998, |
| "epoch": 0.7372505543237251, |
| "frac_reward_zero_std": 0.54375, |
| "grad_norm": 0.119140625, |
| "learning_rate": 9.8671e-06, |
| "loss": -0.0253, |
| "num_tokens": 128377922.0, |
| "reward": 0.4149218946695328, |
| "reward_std": 0.4549461007118225, |
| "rewards/reward_accuracy/mean": 0.31640625, |
| "rewards/reward_accuracy/std": 0.4538107842206955, |
| "rewards/reward_format/mean": 0.09851562678813934, |
| "rewards/reward_format/std": 0.010208774264901877, |
| "sampling/importance_sampling_ratio/max": 2.71019971370697, |
| "sampling/importance_sampling_ratio/mean": 0.8159892857074738, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.699064952135086, |
| "sampling/sampling_logp_difference/mean": 0.009595869202166795, |
| "step": 1330, |
| "step_time": 25.558494247263297 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2713.8, |
| "completions/max_terminated_length": 1954.4, |
| "completions/mean_length": 513.18984375, |
| "completions/mean_terminated_length": 493.2311492919922, |
| "completions/min_length": 85.4, |
| "completions/min_terminated_length": 85.4, |
| "entropy": 0.176772026065737, |
| "epoch": 0.7427937915742794, |
| "frac_reward_zero_std": 0.49375, |
| "grad_norm": 0.236328125, |
| "learning_rate": 9.8661e-06, |
| "loss": -0.0035, |
| "num_tokens": 129342093.0, |
| "reward": 0.43683595657348634, |
| "reward_std": 0.4683630168437958, |
| "rewards/reward_accuracy/mean": 0.33828125, |
| "rewards/reward_accuracy/std": 0.46728394329547884, |
| "rewards/reward_format/mean": 0.09855469018220901, |
| "rewards/reward_format/std": 0.009797102957963943, |
| "sampling/importance_sampling_ratio/max": 2.8915267467498778, |
| "sampling/importance_sampling_ratio/mean": 0.8373220384120941, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7309032797813415, |
| "sampling/sampling_logp_difference/mean": 0.009342831280082464, |
| "step": 1340, |
| "step_time": 25.362048085965217 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2669.8, |
| "completions/max_terminated_length": 2021.8, |
| "completions/mean_length": 571.31796875, |
| "completions/mean_terminated_length": 551.9554718017578, |
| "completions/min_length": 110.4, |
| "completions/min_terminated_length": 110.4, |
| "entropy": 0.16863044598139823, |
| "epoch": 0.7483370288248337, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.2216796875, |
| "learning_rate": 9.865100000000001e-06, |
| "loss": 0.0095, |
| "num_tokens": 130382468.0, |
| "reward": 0.32253907322883607, |
| "reward_std": 0.39809717535972594, |
| "rewards/reward_accuracy/mean": 0.22421875, |
| "rewards/reward_accuracy/std": 0.3979575902223587, |
| "rewards/reward_format/mean": 0.09832031354308128, |
| "rewards/reward_format/std": 0.010266939364373683, |
| "sampling/importance_sampling_ratio/max": 2.718535327911377, |
| "sampling/importance_sampling_ratio/mean": 0.841222733259201, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.2803695440292358, |
| "sampling/sampling_logp_difference/mean": 0.008856746926903724, |
| "step": 1350, |
| "step_time": 25.16195757430978 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0125, |
| "completions/max_length": 2822.6, |
| "completions/max_terminated_length": 1913.9, |
| "completions/mean_length": 553.975, |
| "completions/mean_terminated_length": 522.246630859375, |
| "completions/min_length": 108.0, |
| "completions/min_terminated_length": 108.0, |
| "entropy": 0.17616150537505745, |
| "epoch": 0.753880266075388, |
| "frac_reward_zero_std": 0.50625, |
| "grad_norm": 0.29296875, |
| "learning_rate": 9.864100000000001e-06, |
| "loss": -0.0087, |
| "num_tokens": 131398244.0, |
| "reward": 0.35628907978534696, |
| "reward_std": 0.4342294007539749, |
| "rewards/reward_accuracy/mean": 0.25859375, |
| "rewards/reward_accuracy/std": 0.4326260656118393, |
| "rewards/reward_format/mean": 0.09769531264901161, |
| "rewards/reward_format/std": 0.01314905546605587, |
| "sampling/importance_sampling_ratio/max": 2.8372369527816774, |
| "sampling/importance_sampling_ratio/mean": 0.789298415184021, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7425859689712524, |
| "sampling/sampling_logp_difference/mean": 0.009495837520807982, |
| "step": 1360, |
| "step_time": 26.573665570514276 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2374.6, |
| "completions/max_terminated_length": 1960.6, |
| "completions/mean_length": 530.14296875, |
| "completions/mean_terminated_length": 512.4293518066406, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.17306726663373412, |
| "epoch": 0.7594235033259423, |
| "frac_reward_zero_std": 0.60625, |
| "grad_norm": 0.2333984375, |
| "learning_rate": 9.8631e-06, |
| "loss": -0.0205, |
| "num_tokens": 132381979.0, |
| "reward": 0.39328126758337023, |
| "reward_std": 0.4454226493835449, |
| "rewards/reward_accuracy/mean": 0.29453125, |
| "rewards/reward_accuracy/std": 0.4449634224176407, |
| "rewards/reward_format/mean": 0.09875000044703483, |
| "rewards/reward_format/std": 0.008807667205110193, |
| "sampling/importance_sampling_ratio/max": 2.814861226081848, |
| "sampling/importance_sampling_ratio/mean": 0.7846118390560151, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7451088309288025, |
| "sampling/sampling_logp_difference/mean": 0.009118443727493286, |
| "step": 1370, |
| "step_time": 22.149936430342496 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2600.1, |
| "completions/max_terminated_length": 1838.2, |
| "completions/mean_length": 540.10390625, |
| "completions/mean_terminated_length": 512.0678283691407, |
| "completions/min_length": 105.3, |
| "completions/min_terminated_length": 105.3, |
| "entropy": 0.1879195018671453, |
| "epoch": 0.7649667405764967, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.24609375, |
| "learning_rate": 9.862100000000002e-06, |
| "loss": -0.0081, |
| "num_tokens": 133370728.0, |
| "reward": 0.352187517285347, |
| "reward_std": 0.4320010393857956, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4307866275310516, |
| "rewards/reward_format/mean": 0.09828125312924385, |
| "rewards/reward_format/std": 0.011587123712524771, |
| "sampling/importance_sampling_ratio/max": 2.7622841119766237, |
| "sampling/importance_sampling_ratio/mean": 0.792380976676941, |
| "sampling/importance_sampling_ratio/min": 0.0005922972690314054, |
| "sampling/sampling_logp_difference/max": 0.6667560577392578, |
| "sampling/sampling_logp_difference/mean": 0.009955601207911969, |
| "step": 1380, |
| "step_time": 24.284529462317003 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.003125, |
| "completions/max_length": 2058.6, |
| "completions/max_terminated_length": 1716.6, |
| "completions/mean_length": 514.1921875, |
| "completions/mean_terminated_length": 506.3177978515625, |
| "completions/min_length": 93.4, |
| "completions/min_terminated_length": 93.4, |
| "entropy": 0.16732011446729303, |
| "epoch": 0.770509977827051, |
| "frac_reward_zero_std": 0.6625, |
| "grad_norm": 0.0458984375, |
| "learning_rate": 9.861100000000001e-06, |
| "loss": -0.0305, |
| "num_tokens": 134334310.0, |
| "reward": 0.3969140827655792, |
| "reward_std": 0.4520221710205078, |
| "rewards/reward_accuracy/mean": 0.29765625, |
| "rewards/reward_accuracy/std": 0.4518384009599686, |
| "rewards/reward_format/mean": 0.0992578148841858, |
| "rewards/reward_format/std": 0.005298975668847561, |
| "sampling/importance_sampling_ratio/max": 2.895813465118408, |
| "sampling/importance_sampling_ratio/mean": 0.8245405972003936, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7564067602157593, |
| "sampling/sampling_logp_difference/mean": 0.00908391186967492, |
| "step": 1390, |
| "step_time": 19.907496875617653 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2754.3, |
| "completions/max_terminated_length": 1808.0, |
| "completions/mean_length": 552.26640625, |
| "completions/mean_terminated_length": 528.4343017578125, |
| "completions/min_length": 96.9, |
| "completions/min_terminated_length": 96.9, |
| "entropy": 0.16637535467743875, |
| "epoch": 0.7760532150776053, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.07861328125, |
| "learning_rate": 9.8601e-06, |
| "loss": -0.0073, |
| "num_tokens": 135340563.0, |
| "reward": 0.36046877801418303, |
| "reward_std": 0.43140116333961487, |
| "rewards/reward_accuracy/mean": 0.26171875, |
| "rewards/reward_accuracy/std": 0.4307568043470383, |
| "rewards/reward_format/mean": 0.09875000342726707, |
| "rewards/reward_format/std": 0.00973450131714344, |
| "sampling/importance_sampling_ratio/max": 2.7701124429702757, |
| "sampling/importance_sampling_ratio/mean": 0.8139499366283417, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6398482322692871, |
| "sampling/sampling_logp_difference/mean": 0.008860704116523265, |
| "step": 1400, |
| "step_time": 25.946086296066643 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0125, |
| "completions/max_length": 2914.4, |
| "completions/max_terminated_length": 1988.7, |
| "completions/mean_length": 547.91875, |
| "completions/mean_terminated_length": 515.975537109375, |
| "completions/min_length": 105.6, |
| "completions/min_terminated_length": 105.6, |
| "entropy": 0.16786429462954403, |
| "epoch": 0.7815964523281597, |
| "frac_reward_zero_std": 0.5875, |
| "grad_norm": 0.2060546875, |
| "learning_rate": 9.8591e-06, |
| "loss": -0.0174, |
| "num_tokens": 136353899.0, |
| "reward": 0.33253908157348633, |
| "reward_std": 0.4233589291572571, |
| "rewards/reward_accuracy/mean": 0.234375, |
| "rewards/reward_accuracy/std": 0.4221883028745651, |
| "rewards/reward_format/mean": 0.09816406518220902, |
| "rewards/reward_format/std": 0.011593225318938494, |
| "sampling/importance_sampling_ratio/max": 2.7559324979782103, |
| "sampling/importance_sampling_ratio/mean": 0.8059645771980286, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.9507427930831909, |
| "sampling/sampling_logp_difference/mean": 0.008886136207729578, |
| "step": 1410, |
| "step_time": 27.101147025637328 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2615.2, |
| "completions/max_terminated_length": 2087.8, |
| "completions/mean_length": 485.61015625, |
| "completions/mean_terminated_length": 461.0500854492187, |
| "completions/min_length": 72.2, |
| "completions/min_terminated_length": 72.2, |
| "entropy": 0.1703641911968589, |
| "epoch": 0.7871396895787139, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.2333984375, |
| "learning_rate": 9.8581e-06, |
| "loss": -0.0038, |
| "num_tokens": 137269648.0, |
| "reward": 0.432109397649765, |
| "reward_std": 0.4636314153671265, |
| "rewards/reward_accuracy/mean": 0.33359375, |
| "rewards/reward_accuracy/std": 0.46261965334415434, |
| "rewards/reward_format/mean": 0.09851562827825547, |
| "rewards/reward_format/std": 0.00931641119532287, |
| "sampling/importance_sampling_ratio/max": 2.7701186180114745, |
| "sampling/importance_sampling_ratio/mean": 0.8541751205921173, |
| "sampling/importance_sampling_ratio/min": 0.003853759169578552, |
| "sampling/sampling_logp_difference/max": 0.6677386462688446, |
| "sampling/sampling_logp_difference/mean": 0.008948878198862076, |
| "step": 1420, |
| "step_time": 24.462557940883563 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2827.6, |
| "completions/max_terminated_length": 2089.7, |
| "completions/mean_length": 523.37890625, |
| "completions/mean_terminated_length": 493.2789367675781, |
| "completions/min_length": 100.3, |
| "completions/min_terminated_length": 100.3, |
| "entropy": 0.17284738696180285, |
| "epoch": 0.7926829268292683, |
| "frac_reward_zero_std": 0.625, |
| "grad_norm": 0.185546875, |
| "learning_rate": 9.857100000000001e-06, |
| "loss": -0.0078, |
| "num_tokens": 138236613.0, |
| "reward": 0.3709375157952309, |
| "reward_std": 0.42892016768455504, |
| "rewards/reward_accuracy/mean": 0.27265625, |
| "rewards/reward_accuracy/std": 0.4280405670404434, |
| "rewards/reward_format/mean": 0.09828125089406967, |
| "rewards/reward_format/std": 0.010424401424825192, |
| "sampling/importance_sampling_ratio/max": 2.7507201194763184, |
| "sampling/importance_sampling_ratio/mean": 0.812669825553894, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6518543839454651, |
| "sampling/sampling_logp_difference/mean": 0.009192117303609849, |
| "step": 1430, |
| "step_time": 26.481634673988445 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0125, |
| "completions/max_length": 3060.8, |
| "completions/max_terminated_length": 2264.3, |
| "completions/mean_length": 547.78359375, |
| "completions/mean_terminated_length": 515.7974578857422, |
| "completions/min_length": 83.5, |
| "completions/min_terminated_length": 83.5, |
| "entropy": 0.1719004947692156, |
| "epoch": 0.7982261640798226, |
| "frac_reward_zero_std": 0.56875, |
| "grad_norm": 0.294921875, |
| "learning_rate": 9.8561e-06, |
| "loss": -0.0127, |
| "num_tokens": 139233408.0, |
| "reward": 0.41378907710313795, |
| "reward_std": 0.4491315305233002, |
| "rewards/reward_accuracy/mean": 0.315625, |
| "rewards/reward_accuracy/std": 0.44791861772537234, |
| "rewards/reward_format/mean": 0.09816406294703484, |
| "rewards/reward_format/std": 0.011280422657728195, |
| "sampling/importance_sampling_ratio/max": 2.6995941162109376, |
| "sampling/importance_sampling_ratio/mean": 0.8385890603065491, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7107508301734924, |
| "sampling/sampling_logp_difference/mean": 0.008985796011984348, |
| "step": 1440, |
| "step_time": 29.045003307564183 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0125, |
| "completions/max_length": 2791.0, |
| "completions/max_terminated_length": 2069.8, |
| "completions/mean_length": 559.42890625, |
| "completions/mean_terminated_length": 528.3026062011719, |
| "completions/min_length": 83.8, |
| "completions/min_terminated_length": 83.8, |
| "entropy": 0.16939815212972462, |
| "epoch": 0.8037694013303769, |
| "frac_reward_zero_std": 0.55, |
| "grad_norm": 0.29296875, |
| "learning_rate": 9.855100000000002e-06, |
| "loss": -0.0128, |
| "num_tokens": 140250853.0, |
| "reward": 0.3386718899011612, |
| "reward_std": 0.4221374124288559, |
| "rewards/reward_accuracy/mean": 0.240625, |
| "rewards/reward_accuracy/std": 0.4208660662174225, |
| "rewards/reward_format/mean": 0.09804687723517418, |
| "rewards/reward_format/std": 0.011670124670490622, |
| "sampling/importance_sampling_ratio/max": 2.746728038787842, |
| "sampling/importance_sampling_ratio/mean": 0.8298609793186188, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6871681630611419, |
| "sampling/sampling_logp_difference/mean": 0.008777304086834192, |
| "step": 1450, |
| "step_time": 25.744083932694046 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2365.6, |
| "completions/max_terminated_length": 1765.4, |
| "completions/mean_length": 499.6296875, |
| "completions/mean_terminated_length": 481.4453857421875, |
| "completions/min_length": 80.8, |
| "completions/min_terminated_length": 80.8, |
| "entropy": 0.1693771872203797, |
| "epoch": 0.8093126385809313, |
| "frac_reward_zero_std": 0.675, |
| "grad_norm": 0.150390625, |
| "learning_rate": 9.854100000000001e-06, |
| "loss": 0.0113, |
| "num_tokens": 141193739.0, |
| "reward": 0.36628908216953276, |
| "reward_std": 0.434375461935997, |
| "rewards/reward_accuracy/mean": 0.2671875, |
| "rewards/reward_accuracy/std": 0.4338067263364792, |
| "rewards/reward_format/mean": 0.0991015650331974, |
| "rewards/reward_format/std": 0.005819725338369608, |
| "sampling/importance_sampling_ratio/max": 2.8382567882537844, |
| "sampling/importance_sampling_ratio/mean": 0.8534024000167847, |
| "sampling/importance_sampling_ratio/min": 0.006124608963727951, |
| "sampling/sampling_logp_difference/max": 0.7115189671516419, |
| "sampling/sampling_logp_difference/mean": 0.009044642280787229, |
| "step": 1460, |
| "step_time": 22.214779848000035 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2714.8, |
| "completions/max_terminated_length": 1916.9, |
| "completions/mean_length": 528.03125, |
| "completions/mean_terminated_length": 500.15110778808594, |
| "completions/min_length": 96.3, |
| "completions/min_terminated_length": 96.3, |
| "entropy": 0.1756370748858899, |
| "epoch": 0.8148558758314856, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.1728515625, |
| "learning_rate": 9.8531e-06, |
| "loss": -0.0136, |
| "num_tokens": 142166091.0, |
| "reward": 0.353984397649765, |
| "reward_std": 0.43734304904937743, |
| "rewards/reward_accuracy/mean": 0.25546875, |
| "rewards/reward_accuracy/std": 0.43633460998535156, |
| "rewards/reward_format/mean": 0.09851562678813934, |
| "rewards/reward_format/std": 0.010675186477601527, |
| "sampling/importance_sampling_ratio/max": 2.740734505653381, |
| "sampling/importance_sampling_ratio/mean": 0.8083930075168609, |
| "sampling/importance_sampling_ratio/min": 0.003525478392839432, |
| "sampling/sampling_logp_difference/max": 0.8070775091648101, |
| "sampling/sampling_logp_difference/mean": 0.009227780997753144, |
| "step": 1470, |
| "step_time": 25.408219558466225 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2629.4, |
| "completions/max_terminated_length": 1907.0, |
| "completions/mean_length": 517.4328125, |
| "completions/mean_terminated_length": 489.47357482910155, |
| "completions/min_length": 83.3, |
| "completions/min_terminated_length": 83.3, |
| "entropy": 0.16952012055553495, |
| "epoch": 0.8203991130820399, |
| "frac_reward_zero_std": 0.5375, |
| "grad_norm": 0.32421875, |
| "learning_rate": 9.852100000000002e-06, |
| "loss": 0.0164, |
| "num_tokens": 143136797.0, |
| "reward": 0.3515625298023224, |
| "reward_std": 0.4281261473894119, |
| "rewards/reward_accuracy/mean": 0.253125, |
| "rewards/reward_accuracy/std": 0.4270993769168854, |
| "rewards/reward_format/mean": 0.09843750298023224, |
| "rewards/reward_format/std": 0.010326443100348115, |
| "sampling/importance_sampling_ratio/max": 2.755756139755249, |
| "sampling/importance_sampling_ratio/mean": 0.8674587726593017, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7074837744235992, |
| "sampling/sampling_logp_difference/mean": 0.008972132112830877, |
| "step": 1480, |
| "step_time": 24.8666845722124 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2397.9, |
| "completions/max_terminated_length": 1771.2, |
| "completions/mean_length": 496.1765625, |
| "completions/mean_terminated_length": 480.33441772460935, |
| "completions/min_length": 86.4, |
| "completions/min_terminated_length": 86.4, |
| "entropy": 0.17569130258634685, |
| "epoch": 0.8259423503325942, |
| "frac_reward_zero_std": 0.6125, |
| "grad_norm": 0.146484375, |
| "learning_rate": 9.851100000000001e-06, |
| "loss": 0.0044, |
| "num_tokens": 144074823.0, |
| "reward": 0.37972658723592756, |
| "reward_std": 0.4411593586206436, |
| "rewards/reward_accuracy/mean": 0.28046875, |
| "rewards/reward_accuracy/std": 0.440687894821167, |
| "rewards/reward_format/mean": 0.0992578148841858, |
| "rewards/reward_format/std": 0.005995583906769753, |
| "sampling/importance_sampling_ratio/max": 2.722913146018982, |
| "sampling/importance_sampling_ratio/mean": 0.8369788706302643, |
| "sampling/importance_sampling_ratio/min": 0.004360070824623108, |
| "sampling/sampling_logp_difference/max": 0.7009375095367432, |
| "sampling/sampling_logp_difference/mean": 0.009167639631778001, |
| "step": 1490, |
| "step_time": 22.801207183580846 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2952.4, |
| "completions/max_terminated_length": 1852.9, |
| "completions/mean_length": 480.96171875, |
| "completions/mean_terminated_length": 452.29551696777344, |
| "completions/min_length": 98.8, |
| "completions/min_terminated_length": 98.8, |
| "entropy": 0.1692362554371357, |
| "epoch": 0.8314855875831486, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.1533203125, |
| "learning_rate": 9.8501e-06, |
| "loss": -0.0079, |
| "num_tokens": 144987014.0, |
| "reward": 0.3659765854477882, |
| "reward_std": 0.43042272627353667, |
| "rewards/reward_accuracy/mean": 0.2671875, |
| "rewards/reward_accuracy/std": 0.4295419603586197, |
| "rewards/reward_format/mean": 0.09878906607627869, |
| "rewards/reward_format/std": 0.009815829154103995, |
| "sampling/importance_sampling_ratio/max": 2.8772937059402466, |
| "sampling/importance_sampling_ratio/mean": 0.8501492500305176, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6571451187133789, |
| "sampling/sampling_logp_difference/mean": 0.008971773087978363, |
| "step": 1500, |
| "step_time": 27.21587958172895 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2198.0, |
| "completions/max_terminated_length": 1672.5, |
| "completions/mean_length": 468.16953125, |
| "completions/mean_terminated_length": 448.1154449462891, |
| "completions/min_length": 97.9, |
| "completions/min_terminated_length": 97.9, |
| "entropy": 0.1714822597336024, |
| "epoch": 0.8370288248337029, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.2373046875, |
| "learning_rate": 9.8491e-06, |
| "loss": 0.0001, |
| "num_tokens": 145889303.0, |
| "reward": 0.39746096134185793, |
| "reward_std": 0.4559736341238022, |
| "rewards/reward_accuracy/mean": 0.2984375, |
| "rewards/reward_accuracy/std": 0.4553667098283768, |
| "rewards/reward_format/mean": 0.09902343973517418, |
| "rewards/reward_format/std": 0.006176626868546009, |
| "sampling/importance_sampling_ratio/max": 2.679671359062195, |
| "sampling/importance_sampling_ratio/mean": 0.8520968854427338, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.619900006055832, |
| "sampling/sampling_logp_difference/mean": 0.009144706465303899, |
| "step": 1510, |
| "step_time": 20.706707979179917 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2791.0, |
| "completions/max_terminated_length": 2208.8, |
| "completions/mean_length": 509.10390625, |
| "completions/mean_terminated_length": 484.7663543701172, |
| "completions/min_length": 81.4, |
| "completions/min_terminated_length": 81.4, |
| "entropy": 0.16929919156245887, |
| "epoch": 0.8425720620842572, |
| "frac_reward_zero_std": 0.65, |
| "grad_norm": 0.08349609375, |
| "learning_rate": 9.8481e-06, |
| "loss": -0.0058, |
| "num_tokens": 146834492.0, |
| "reward": 0.4087890863418579, |
| "reward_std": 0.45253988802433015, |
| "rewards/reward_accuracy/mean": 0.31015625, |
| "rewards/reward_accuracy/std": 0.4514791578054428, |
| "rewards/reward_format/mean": 0.09863281473517418, |
| "rewards/reward_format/std": 0.009044339694082738, |
| "sampling/importance_sampling_ratio/max": 2.6706568002700806, |
| "sampling/importance_sampling_ratio/mean": 0.8127996385097503, |
| "sampling/importance_sampling_ratio/min": 0.001011876855045557, |
| "sampling/sampling_logp_difference/max": 0.7441758453845978, |
| "sampling/sampling_logp_difference/mean": 0.008832585811614991, |
| "step": 1520, |
| "step_time": 26.427697759540752 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2677.5, |
| "completions/max_terminated_length": 1895.1, |
| "completions/mean_length": 510.8828125, |
| "completions/mean_terminated_length": 484.76210021972656, |
| "completions/min_length": 89.9, |
| "completions/min_terminated_length": 89.9, |
| "entropy": 0.17572491336613894, |
| "epoch": 0.8481152993348116, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.330078125, |
| "learning_rate": 9.847100000000001e-06, |
| "loss": 0.0055, |
| "num_tokens": 147795894.0, |
| "reward": 0.3181250110268593, |
| "reward_std": 0.406618195772171, |
| "rewards/reward_accuracy/mean": 0.21953125, |
| "rewards/reward_accuracy/std": 0.4057704359292984, |
| "rewards/reward_format/mean": 0.09859375059604644, |
| "rewards/reward_format/std": 0.009827477717772126, |
| "sampling/importance_sampling_ratio/max": 2.697995328903198, |
| "sampling/importance_sampling_ratio/mean": 0.8356251657009125, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6047917127609252, |
| "sampling/sampling_logp_difference/mean": 0.009251200687140226, |
| "step": 1530, |
| "step_time": 25.106494162930176 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2785.5, |
| "completions/max_terminated_length": 1951.9, |
| "completions/mean_length": 487.2140625, |
| "completions/mean_terminated_length": 468.92139892578126, |
| "completions/min_length": 110.3, |
| "completions/min_terminated_length": 110.3, |
| "entropy": 0.17829104266129434, |
| "epoch": 0.8536585365853658, |
| "frac_reward_zero_std": 0.63125, |
| "grad_norm": 0.201171875, |
| "learning_rate": 9.8461e-06, |
| "loss": -0.008, |
| "num_tokens": 148723552.0, |
| "reward": 0.3340625211596489, |
| "reward_std": 0.41313970386981963, |
| "rewards/reward_accuracy/mean": 0.23515625, |
| "rewards/reward_accuracy/std": 0.41247397661209106, |
| "rewards/reward_format/mean": 0.0989062525331974, |
| "rewards/reward_format/std": 0.008803461026400328, |
| "sampling/importance_sampling_ratio/max": 2.7922824382781983, |
| "sampling/importance_sampling_ratio/mean": 0.8355647981166839, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6723552346229553, |
| "sampling/sampling_logp_difference/mean": 0.00926859900355339, |
| "step": 1540, |
| "step_time": 25.238597383489832 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2838.6, |
| "completions/max_terminated_length": 1869.7, |
| "completions/mean_length": 520.75546875, |
| "completions/mean_terminated_length": 495.0293762207031, |
| "completions/min_length": 101.6, |
| "completions/min_terminated_length": 101.6, |
| "entropy": 0.17204268015921115, |
| "epoch": 0.8592017738359202, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.2373046875, |
| "learning_rate": 9.8451e-06, |
| "loss": -0.0158, |
| "num_tokens": 149695895.0, |
| "reward": 0.3524609595537186, |
| "reward_std": 0.4265292227268219, |
| "rewards/reward_accuracy/mean": 0.25390625, |
| "rewards/reward_accuracy/std": 0.4256564646959305, |
| "rewards/reward_format/mean": 0.09855469092726707, |
| "rewards/reward_format/std": 0.009816759079694749, |
| "sampling/importance_sampling_ratio/max": 2.768324613571167, |
| "sampling/importance_sampling_ratio/mean": 0.8370575964450836, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7241550922393799, |
| "sampling/sampling_logp_difference/mean": 0.008993833558633924, |
| "step": 1550, |
| "step_time": 26.116909568291156 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01484375, |
| "completions/max_length": 2859.4, |
| "completions/max_terminated_length": 2310.1, |
| "completions/mean_length": 514.56171875, |
| "completions/mean_terminated_length": 476.2007263183594, |
| "completions/min_length": 91.1, |
| "completions/min_terminated_length": 91.1, |
| "entropy": 0.17101809177547694, |
| "epoch": 0.8647450110864745, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.1201171875, |
| "learning_rate": 9.844100000000001e-06, |
| "loss": -0.0069, |
| "num_tokens": 150667542.0, |
| "reward": 0.371718767285347, |
| "reward_std": 0.4337207764387131, |
| "rewards/reward_accuracy/mean": 0.2734375, |
| "rewards/reward_accuracy/std": 0.432559135556221, |
| "rewards/reward_format/mean": 0.09828125089406967, |
| "rewards/reward_format/std": 0.011193818133324384, |
| "sampling/importance_sampling_ratio/max": 2.7850489377975465, |
| "sampling/importance_sampling_ratio/mean": 0.8497204422950745, |
| "sampling/importance_sampling_ratio/min": 0.0017529217526316642, |
| "sampling/sampling_logp_difference/max": 0.7319178342819214, |
| "sampling/sampling_logp_difference/mean": 0.008720296062529087, |
| "step": 1560, |
| "step_time": 26.913610259536654 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2507.8, |
| "completions/max_terminated_length": 1854.0, |
| "completions/mean_length": 478.31640625, |
| "completions/mean_terminated_length": 456.0650207519531, |
| "completions/min_length": 83.4, |
| "completions/min_terminated_length": 83.4, |
| "entropy": 0.17490067863836883, |
| "epoch": 0.8702882483370288, |
| "frac_reward_zero_std": 0.6375, |
| "grad_norm": 0.2490234375, |
| "learning_rate": 9.8431e-06, |
| "loss": -0.0128, |
| "num_tokens": 151588899.0, |
| "reward": 0.4020312711596489, |
| "reward_std": 0.4310712248086929, |
| "rewards/reward_accuracy/mean": 0.303125, |
| "rewards/reward_accuracy/std": 0.4303927838802338, |
| "rewards/reward_format/mean": 0.09890625178813935, |
| "rewards/reward_format/std": 0.008371657691895962, |
| "sampling/importance_sampling_ratio/max": 2.750538206100464, |
| "sampling/importance_sampling_ratio/mean": 0.8472043991088867, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6552417695522308, |
| "sampling/sampling_logp_difference/mean": 0.009253245126456023, |
| "step": 1570, |
| "step_time": 23.56525327046402 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2756.2, |
| "completions/max_terminated_length": 2024.6, |
| "completions/mean_length": 492.58046875, |
| "completions/mean_terminated_length": 472.2879211425781, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.16003988734446467, |
| "epoch": 0.8758314855875832, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.1474609375, |
| "learning_rate": 9.842100000000002e-06, |
| "loss": -0.0027, |
| "num_tokens": 152519034.0, |
| "reward": 0.428046903014183, |
| "reward_std": 0.46280158162117, |
| "rewards/reward_accuracy/mean": 0.32890625, |
| "rewards/reward_accuracy/std": 0.46209281086921694, |
| "rewards/reward_format/mean": 0.09914062768220902, |
| "rewards/reward_format/std": 0.007443396002054214, |
| "sampling/importance_sampling_ratio/max": 2.794764590263367, |
| "sampling/importance_sampling_ratio/mean": 0.874411141872406, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7365697383880615, |
| "sampling/sampling_logp_difference/mean": 0.008330690767616033, |
| "step": 1580, |
| "step_time": 25.51660825824365 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2635.5, |
| "completions/max_terminated_length": 1962.9, |
| "completions/mean_length": 492.16796875, |
| "completions/mean_terminated_length": 461.50926208496094, |
| "completions/min_length": 99.3, |
| "completions/min_terminated_length": 99.3, |
| "entropy": 0.16928286892361938, |
| "epoch": 0.8813747228381374, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.44140625, |
| "learning_rate": 9.841100000000001e-06, |
| "loss": -0.0196, |
| "num_tokens": 153448793.0, |
| "reward": 0.41492189541459085, |
| "reward_std": 0.4259640336036682, |
| "rewards/reward_accuracy/mean": 0.31640625, |
| "rewards/reward_accuracy/std": 0.42477800622582434, |
| "rewards/reward_format/mean": 0.09851562678813934, |
| "rewards/reward_format/std": 0.009721684735268354, |
| "sampling/importance_sampling_ratio/max": 2.846252179145813, |
| "sampling/importance_sampling_ratio/mean": 0.8855127811431884, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7301100432872772, |
| "sampling/sampling_logp_difference/mean": 0.008910700213164091, |
| "step": 1590, |
| "step_time": 24.648634646087885 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2787.8, |
| "completions/max_terminated_length": 2066.3, |
| "completions/mean_length": 508.634375, |
| "completions/mean_terminated_length": 482.4706970214844, |
| "completions/min_length": 86.9, |
| "completions/min_terminated_length": 86.9, |
| "entropy": 0.16741596949286758, |
| "epoch": 0.8869179600886918, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.443359375, |
| "learning_rate": 9.840100000000001e-06, |
| "loss": 0.0057, |
| "num_tokens": 154394949.0, |
| "reward": 0.40425783544778826, |
| "reward_std": 0.45022012293338776, |
| "rewards/reward_accuracy/mean": 0.30546875, |
| "rewards/reward_accuracy/std": 0.4493494898080826, |
| "rewards/reward_format/mean": 0.09878906533122063, |
| "rewards/reward_format/std": 0.008775639999657869, |
| "sampling/importance_sampling_ratio/max": 2.8200287103652952, |
| "sampling/importance_sampling_ratio/mean": 0.8577162623405457, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.8249281287193299, |
| "sampling/sampling_logp_difference/mean": 0.008735357830300928, |
| "step": 1600, |
| "step_time": 25.96782773421146 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2640.3, |
| "completions/max_terminated_length": 1956.6, |
| "completions/mean_length": 492.35, |
| "completions/mean_terminated_length": 461.8139129638672, |
| "completions/min_length": 107.3, |
| "completions/min_terminated_length": 107.3, |
| "entropy": 0.17509202077053487, |
| "epoch": 0.8924611973392461, |
| "frac_reward_zero_std": 0.625, |
| "grad_norm": 0.1064453125, |
| "learning_rate": 9.8391e-06, |
| "loss": -0.0003, |
| "num_tokens": 155325469.0, |
| "reward": 0.3351562738418579, |
| "reward_std": 0.41576330065727235, |
| "rewards/reward_accuracy/mean": 0.23671875, |
| "rewards/reward_accuracy/std": 0.41488017737865446, |
| "rewards/reward_format/mean": 0.09843750223517418, |
| "rewards/reward_format/std": 0.01045987056568265, |
| "sampling/importance_sampling_ratio/max": 2.7845786094665526, |
| "sampling/importance_sampling_ratio/mean": 0.8718160808086395, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6115598857402802, |
| "sampling/sampling_logp_difference/mean": 0.009081926196813583, |
| "step": 1610, |
| "step_time": 24.84901190875098 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2731.4, |
| "completions/max_terminated_length": 2076.3, |
| "completions/mean_length": 510.02265625, |
| "completions/mean_terminated_length": 494.0275817871094, |
| "completions/min_length": 126.3, |
| "completions/min_terminated_length": 126.3, |
| "entropy": 0.1763566104695201, |
| "epoch": 0.8980044345898004, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 0.3203125, |
| "learning_rate": 9.8381e-06, |
| "loss": -0.0256, |
| "num_tokens": 156277906.0, |
| "reward": 0.37097658812999723, |
| "reward_std": 0.43545143902301786, |
| "rewards/reward_accuracy/mean": 0.271875, |
| "rewards/reward_accuracy/std": 0.43477254211902616, |
| "rewards/reward_format/mean": 0.09910156428813935, |
| "rewards/reward_format/std": 0.007304980885237455, |
| "sampling/importance_sampling_ratio/max": 2.736675834655762, |
| "sampling/importance_sampling_ratio/mean": 0.822258710861206, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6640795946121216, |
| "sampling/sampling_logp_difference/mean": 0.009284504316747188, |
| "step": 1620, |
| "step_time": 25.179024444706737 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 2955.5, |
| "completions/max_terminated_length": 1888.1, |
| "completions/mean_length": 544.4203125, |
| "completions/mean_terminated_length": 504.6833526611328, |
| "completions/min_length": 107.4, |
| "completions/min_terminated_length": 107.4, |
| "entropy": 0.1677720178849995, |
| "epoch": 0.9035476718403548, |
| "frac_reward_zero_std": 0.58125, |
| "grad_norm": 0.2294921875, |
| "learning_rate": 9.837100000000001e-06, |
| "loss": -0.0102, |
| "num_tokens": 157285940.0, |
| "reward": 0.3259765811264515, |
| "reward_std": 0.39875063449144366, |
| "rewards/reward_accuracy/mean": 0.228125, |
| "rewards/reward_accuracy/std": 0.3973326973617077, |
| "rewards/reward_format/mean": 0.09785156473517417, |
| "rewards/reward_format/std": 0.012638452928513288, |
| "sampling/importance_sampling_ratio/max": 2.786471152305603, |
| "sampling/importance_sampling_ratio/mean": 0.8493579447269439, |
| "sampling/importance_sampling_ratio/min": 0.00713333860039711, |
| "sampling/sampling_logp_difference/max": 0.6766576588153839, |
| "sampling/sampling_logp_difference/mean": 0.008765450865030288, |
| "step": 1630, |
| "step_time": 27.59880325756967 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01171875, |
| "completions/max_length": 2804.2, |
| "completions/max_terminated_length": 2005.1, |
| "completions/mean_length": 498.325, |
| "completions/mean_terminated_length": 467.8439178466797, |
| "completions/min_length": 83.2, |
| "completions/min_terminated_length": 83.2, |
| "entropy": 0.16951012928038836, |
| "epoch": 0.9090909090909091, |
| "frac_reward_zero_std": 0.63125, |
| "grad_norm": 0.1513671875, |
| "learning_rate": 9.8361e-06, |
| "loss": 0.009, |
| "num_tokens": 158234156.0, |
| "reward": 0.3285156413912773, |
| "reward_std": 0.39574919641017914, |
| "rewards/reward_accuracy/mean": 0.23046875, |
| "rewards/reward_accuracy/std": 0.3946117013692856, |
| "rewards/reward_format/mean": 0.09804687798023223, |
| "rewards/reward_format/std": 0.01179961534217, |
| "sampling/importance_sampling_ratio/max": 2.670122218132019, |
| "sampling/importance_sampling_ratio/mean": 0.8305561125278473, |
| "sampling/importance_sampling_ratio/min": 0.002316636219620705, |
| "sampling/sampling_logp_difference/max": 0.6718661069869996, |
| "sampling/sampling_logp_difference/mean": 0.008881897758692503, |
| "step": 1640, |
| "step_time": 26.312598433345556 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0109375, |
| "completions/max_length": 2569.7, |
| "completions/max_terminated_length": 1680.4, |
| "completions/mean_length": 460.01953125, |
| "completions/mean_terminated_length": 431.3070526123047, |
| "completions/min_length": 91.2, |
| "completions/min_terminated_length": 91.2, |
| "entropy": 0.1646976341959089, |
| "epoch": 0.9146341463414634, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.267578125, |
| "learning_rate": 9.8351e-06, |
| "loss": -0.0034, |
| "num_tokens": 159127061.0, |
| "reward": 0.4166015774011612, |
| "reward_std": 0.4581416755914688, |
| "rewards/reward_accuracy/mean": 0.31796875, |
| "rewards/reward_accuracy/std": 0.4570658206939697, |
| "rewards/reward_format/mean": 0.09863281399011611, |
| "rewards/reward_format/std": 0.009980941284447908, |
| "sampling/importance_sampling_ratio/max": 2.5590462684631348, |
| "sampling/importance_sampling_ratio/mean": 0.8612177610397339, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6283387005329132, |
| "sampling/sampling_logp_difference/mean": 0.008756402507424354, |
| "step": 1650, |
| "step_time": 24.2002768763341 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2329.8, |
| "completions/max_terminated_length": 1852.4, |
| "completions/mean_length": 452.08359375, |
| "completions/mean_terminated_length": 441.85455017089845, |
| "completions/min_length": 75.5, |
| "completions/min_terminated_length": 75.5, |
| "entropy": 0.17603549668565394, |
| "epoch": 0.9201773835920177, |
| "frac_reward_zero_std": 0.675, |
| "grad_norm": 0.18359375, |
| "learning_rate": 9.834100000000001e-06, |
| "loss": 0.0041, |
| "num_tokens": 160006376.0, |
| "reward": 0.3525000214576721, |
| "reward_std": 0.42779700458049774, |
| "rewards/reward_accuracy/mean": 0.253125, |
| "rewards/reward_accuracy/std": 0.42746289670467374, |
| "rewards/reward_format/mean": 0.09937500134110451, |
| "rewards/reward_format/std": 0.0058371412567794325, |
| "sampling/importance_sampling_ratio/max": 2.7684380054473876, |
| "sampling/importance_sampling_ratio/mean": 0.879950761795044, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.9362735986709595, |
| "sampling/sampling_logp_difference/mean": 0.009393100161105394, |
| "step": 1660, |
| "step_time": 21.74191816165112 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2224.4, |
| "completions/max_terminated_length": 1748.0, |
| "completions/mean_length": 492.7375, |
| "completions/mean_terminated_length": 478.5474334716797, |
| "completions/min_length": 87.5, |
| "completions/min_terminated_length": 87.5, |
| "entropy": 0.17344923284836114, |
| "epoch": 0.9257206208425721, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.15234375, |
| "learning_rate": 9.8331e-06, |
| "loss": 0.004, |
| "num_tokens": 160942408.0, |
| "reward": 0.32000001668930056, |
| "reward_std": 0.41126050651073454, |
| "rewards/reward_accuracy/mean": 0.22109375, |
| "rewards/reward_accuracy/std": 0.4105852574110031, |
| "rewards/reward_format/mean": 0.09890625178813935, |
| "rewards/reward_format/std": 0.0067259506322443485, |
| "sampling/importance_sampling_ratio/max": 2.6697850465774535, |
| "sampling/importance_sampling_ratio/mean": 0.8652226269245148, |
| "sampling/importance_sampling_ratio/min": 0.0018840473145246505, |
| "sampling/sampling_logp_difference/max": 0.6511864781379699, |
| "sampling/sampling_logp_difference/mean": 0.008929469622671604, |
| "step": 1670, |
| "step_time": 21.23895557024516 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2211.0, |
| "completions/max_terminated_length": 1852.1, |
| "completions/mean_length": 490.76015625, |
| "completions/mean_terminated_length": 478.92136535644534, |
| "completions/min_length": 103.2, |
| "completions/min_terminated_length": 103.2, |
| "entropy": 0.16392312892712652, |
| "epoch": 0.9312638580931264, |
| "frac_reward_zero_std": 0.6875, |
| "grad_norm": 0.20703125, |
| "learning_rate": 9.8321e-06, |
| "loss": 0.0038, |
| "num_tokens": 161874293.0, |
| "reward": 0.346054707467556, |
| "reward_std": 0.42307898998260496, |
| "rewards/reward_accuracy/mean": 0.246875, |
| "rewards/reward_accuracy/std": 0.4225316524505615, |
| "rewards/reward_format/mean": 0.09917969033122062, |
| "rewards/reward_format/std": 0.006971687171608209, |
| "sampling/importance_sampling_ratio/max": 2.718043637275696, |
| "sampling/importance_sampling_ratio/mean": 0.8333092451095581, |
| "sampling/importance_sampling_ratio/min": 0.01002514585852623, |
| "sampling/sampling_logp_difference/max": 0.7573684632778168, |
| "sampling/sampling_logp_difference/mean": 0.008667673356831074, |
| "step": 1680, |
| "step_time": 20.85160843920894 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2691.9, |
| "completions/max_terminated_length": 1902.0, |
| "completions/mean_length": 489.70625, |
| "completions/mean_terminated_length": 467.40928649902344, |
| "completions/min_length": 94.3, |
| "completions/min_terminated_length": 94.3, |
| "entropy": 0.17194245737046004, |
| "epoch": 0.9368070953436807, |
| "frac_reward_zero_std": 0.49375, |
| "grad_norm": 0.265625, |
| "learning_rate": 9.831100000000001e-06, |
| "loss": -0.0171, |
| "num_tokens": 162797949.0, |
| "reward": 0.3853906482458115, |
| "reward_std": 0.44079620242118833, |
| "rewards/reward_accuracy/mean": 0.28671875, |
| "rewards/reward_accuracy/std": 0.43992804884910586, |
| "rewards/reward_format/mean": 0.09867187738418579, |
| "rewards/reward_format/std": 0.009708312805742025, |
| "sampling/importance_sampling_ratio/max": 2.7648481845855715, |
| "sampling/importance_sampling_ratio/mean": 0.8446653246879577, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6693904042243958, |
| "sampling/sampling_logp_difference/mean": 0.008960756193846463, |
| "step": 1690, |
| "step_time": 24.67778542698361 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2278.6, |
| "completions/max_terminated_length": 2047.4, |
| "completions/mean_length": 506.14921875, |
| "completions/mean_terminated_length": 491.92356262207034, |
| "completions/min_length": 116.8, |
| "completions/min_terminated_length": 116.8, |
| "entropy": 0.16650078673847019, |
| "epoch": 0.9423503325942351, |
| "frac_reward_zero_std": 0.63125, |
| "grad_norm": 0.0751953125, |
| "learning_rate": 9.830100000000001e-06, |
| "loss": -0.0138, |
| "num_tokens": 163750804.0, |
| "reward": 0.34578127413988113, |
| "reward_std": 0.4263346612453461, |
| "rewards/reward_accuracy/mean": 0.246875, |
| "rewards/reward_accuracy/std": 0.42559804022312164, |
| "rewards/reward_format/mean": 0.09890625104308129, |
| "rewards/reward_format/std": 0.00704873614013195, |
| "sampling/importance_sampling_ratio/max": 2.821058964729309, |
| "sampling/importance_sampling_ratio/mean": 0.8440196871757507, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6771661043167114, |
| "sampling/sampling_logp_difference/mean": 0.008594464045017958, |
| "step": 1700, |
| "step_time": 21.736224101111294 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 1976.8, |
| "completions/max_terminated_length": 1818.9, |
| "completions/mean_length": 503.0640625, |
| "completions/mean_terminated_length": 498.95896911621094, |
| "completions/min_length": 111.2, |
| "completions/min_terminated_length": 111.2, |
| "entropy": 0.1644074387382716, |
| "epoch": 0.9478935698447893, |
| "frac_reward_zero_std": 0.71875, |
| "grad_norm": 0.099609375, |
| "learning_rate": 9.8291e-06, |
| "loss": 0.0118, |
| "num_tokens": 164703470.0, |
| "reward": 0.3542578399181366, |
| "reward_std": 0.4226366519927979, |
| "rewards/reward_accuracy/mean": 0.2546875, |
| "rewards/reward_accuracy/std": 0.42244796454906464, |
| "rewards/reward_format/mean": 0.09957031533122063, |
| "rewards/reward_format/std": 0.0043386613018810746, |
| "sampling/importance_sampling_ratio/max": 2.7857895851135255, |
| "sampling/importance_sampling_ratio/mean": 0.8671269536018371, |
| "sampling/importance_sampling_ratio/min": 0.010579676181077958, |
| "sampling/sampling_logp_difference/max": 0.5834326326847077, |
| "sampling/sampling_logp_difference/mean": 0.00877710459753871, |
| "step": 1710, |
| "step_time": 18.886321496451274 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2601.0, |
| "completions/max_terminated_length": 1857.8, |
| "completions/mean_length": 489.38203125, |
| "completions/mean_terminated_length": 469.0405029296875, |
| "completions/min_length": 108.2, |
| "completions/min_terminated_length": 108.2, |
| "entropy": 0.16650945954024793, |
| "epoch": 0.9534368070953437, |
| "frac_reward_zero_std": 0.525, |
| "grad_norm": 0.208984375, |
| "learning_rate": 9.8281e-06, |
| "loss": -0.0202, |
| "num_tokens": 165628367.0, |
| "reward": 0.42250003218650817, |
| "reward_std": 0.46081688106060026, |
| "rewards/reward_accuracy/mean": 0.3234375, |
| "rewards/reward_accuracy/std": 0.46004771888256074, |
| "rewards/reward_format/mean": 0.09906250312924385, |
| "rewards/reward_format/std": 0.008145312312990427, |
| "sampling/importance_sampling_ratio/max": 2.7341848850250243, |
| "sampling/importance_sampling_ratio/mean": 0.8496138572692871, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7799319267272949, |
| "sampling/sampling_logp_difference/mean": 0.008691117353737354, |
| "step": 1720, |
| "step_time": 24.18483052914962 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2349.1, |
| "completions/max_terminated_length": 1791.4, |
| "completions/mean_length": 504.85234375, |
| "completions/mean_terminated_length": 490.94508666992186, |
| "completions/min_length": 90.5, |
| "completions/min_terminated_length": 90.5, |
| "entropy": 0.16250052275136112, |
| "epoch": 0.958980044345898, |
| "frac_reward_zero_std": 0.64375, |
| "grad_norm": 0.296875, |
| "learning_rate": 9.827100000000001e-06, |
| "loss": 0.0125, |
| "num_tokens": 166581226.0, |
| "reward": 0.3507031425833702, |
| "reward_std": 0.41139201670885084, |
| "rewards/reward_accuracy/mean": 0.2515625, |
| "rewards/reward_accuracy/std": 0.41082980781793593, |
| "rewards/reward_format/mean": 0.09914062619209289, |
| "rewards/reward_format/std": 0.006346754170954228, |
| "sampling/importance_sampling_ratio/max": 2.803511714935303, |
| "sampling/importance_sampling_ratio/mean": 0.8566385209560394, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7938909292221069, |
| "sampling/sampling_logp_difference/mean": 0.008522332273423671, |
| "step": 1730, |
| "step_time": 22.460664682276548 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 2754.3, |
| "completions/max_terminated_length": 1978.7, |
| "completions/mean_length": 528.53359375, |
| "completions/mean_terminated_length": 488.36936950683594, |
| "completions/min_length": 99.5, |
| "completions/min_terminated_length": 99.5, |
| "entropy": 0.16177429175004363, |
| "epoch": 0.9645232815964523, |
| "frac_reward_zero_std": 0.56875, |
| "grad_norm": 0.1796875, |
| "learning_rate": 9.8261e-06, |
| "loss": -0.0211, |
| "num_tokens": 167557861.0, |
| "reward": 0.3424609571695328, |
| "reward_std": 0.42351600527763367, |
| "rewards/reward_accuracy/mean": 0.24453125, |
| "rewards/reward_accuracy/std": 0.42236388027668, |
| "rewards/reward_format/mean": 0.09792968928813935, |
| "rewards/reward_format/std": 0.012726074922829867, |
| "sampling/importance_sampling_ratio/max": 2.7701199769973757, |
| "sampling/importance_sampling_ratio/mean": 0.8743699193000793, |
| "sampling/importance_sampling_ratio/min": 0.007349083572626114, |
| "sampling/sampling_logp_difference/max": 0.6878671467304229, |
| "sampling/sampling_logp_difference/mean": 0.008531273901462555, |
| "step": 1740, |
| "step_time": 26.158832946419714 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2585.9, |
| "completions/max_terminated_length": 1859.8, |
| "completions/mean_length": 480.63828125, |
| "completions/mean_terminated_length": 454.2491516113281, |
| "completions/min_length": 77.8, |
| "completions/min_terminated_length": 77.8, |
| "entropy": 0.1643215955235064, |
| "epoch": 0.9700665188470067, |
| "frac_reward_zero_std": 0.725, |
| "grad_norm": 0.12158203125, |
| "learning_rate": 9.8251e-06, |
| "loss": -0.0041, |
| "num_tokens": 168474230.0, |
| "reward": 0.3346484526991844, |
| "reward_std": 0.40806749165058137, |
| "rewards/reward_accuracy/mean": 0.2359375, |
| "rewards/reward_accuracy/std": 0.4074328988790512, |
| "rewards/reward_format/mean": 0.09871093928813934, |
| "rewards/reward_format/std": 0.00881675141863525, |
| "sampling/importance_sampling_ratio/max": 2.6678964853286744, |
| "sampling/importance_sampling_ratio/mean": 0.8714227735996246, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6161976218223572, |
| "sampling/sampling_logp_difference/mean": 0.008625926775857806, |
| "step": 1750, |
| "step_time": 24.627392724109814 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2487.7, |
| "completions/max_terminated_length": 2077.5, |
| "completions/mean_length": 493.55390625, |
| "completions/mean_terminated_length": 473.4448944091797, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.16603692835196854, |
| "epoch": 0.975609756097561, |
| "frac_reward_zero_std": 0.625, |
| "grad_norm": 0.162109375, |
| "learning_rate": 9.824100000000001e-06, |
| "loss": -0.0016, |
| "num_tokens": 169403643.0, |
| "reward": 0.37644533812999725, |
| "reward_std": 0.43830510377883913, |
| "rewards/reward_accuracy/mean": 0.27734375, |
| "rewards/reward_accuracy/std": 0.43764210641384127, |
| "rewards/reward_format/mean": 0.0991015650331974, |
| "rewards/reward_format/std": 0.007093246746808291, |
| "sampling/importance_sampling_ratio/max": 2.743368148803711, |
| "sampling/importance_sampling_ratio/mean": 0.8407061159610748, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7096605896949768, |
| "sampling/sampling_logp_difference/mean": 0.008638298232108354, |
| "step": 1760, |
| "step_time": 23.271483840420842 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2797.1, |
| "completions/max_terminated_length": 2122.4, |
| "completions/mean_length": 501.5671875, |
| "completions/mean_terminated_length": 481.4468566894531, |
| "completions/min_length": 100.1, |
| "completions/min_terminated_length": 100.1, |
| "entropy": 0.16928782761096955, |
| "epoch": 0.9811529933481153, |
| "frac_reward_zero_std": 0.61875, |
| "grad_norm": 0.1083984375, |
| "learning_rate": 9.823100000000001e-06, |
| "loss": 0.0037, |
| "num_tokens": 170341945.0, |
| "reward": 0.3616015911102295, |
| "reward_std": 0.4364154875278473, |
| "rewards/reward_accuracy/mean": 0.2625, |
| "rewards/reward_accuracy/std": 0.43585641086101534, |
| "rewards/reward_format/mean": 0.09910156726837158, |
| "rewards/reward_format/std": 0.008517184667289257, |
| "sampling/importance_sampling_ratio/max": 2.774775576591492, |
| "sampling/importance_sampling_ratio/mean": 0.8304492235183716, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7405862033367157, |
| "sampling/sampling_logp_difference/mean": 0.008952831383794546, |
| "step": 1770, |
| "step_time": 25.67375982780941 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2719.3, |
| "completions/max_terminated_length": 1782.2, |
| "completions/mean_length": 481.01328125, |
| "completions/mean_terminated_length": 464.8302947998047, |
| "completions/min_length": 86.1, |
| "completions/min_terminated_length": 86.1, |
| "entropy": 0.15695817628875375, |
| "epoch": 0.9866962305986696, |
| "frac_reward_zero_std": 0.675, |
| "grad_norm": 0.29296875, |
| "learning_rate": 9.8221e-06, |
| "loss": -0.0007, |
| "num_tokens": 171257474.0, |
| "reward": 0.3444140821695328, |
| "reward_std": 0.4229599803686142, |
| "rewards/reward_accuracy/mean": 0.2453125, |
| "rewards/reward_accuracy/std": 0.42235446870327, |
| "rewards/reward_format/mean": 0.09910156577825546, |
| "rewards/reward_format/std": 0.008056404534727335, |
| "sampling/importance_sampling_ratio/max": 2.6470258474349975, |
| "sampling/importance_sampling_ratio/mean": 0.8684496641159057, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 1.494177520275116, |
| "sampling/sampling_logp_difference/mean": 0.00815441608428955, |
| "step": 1780, |
| "step_time": 24.934749345807358 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00546875, |
| "completions/max_length": 2253.0, |
| "completions/max_terminated_length": 1660.8, |
| "completions/mean_length": 476.815625, |
| "completions/mean_terminated_length": 462.5858154296875, |
| "completions/min_length": 103.2, |
| "completions/min_terminated_length": 103.2, |
| "entropy": 0.16374013400636614, |
| "epoch": 0.9922394678492239, |
| "frac_reward_zero_std": 0.6625, |
| "grad_norm": 0.259765625, |
| "learning_rate": 9.821100000000002e-06, |
| "loss": -0.0033, |
| "num_tokens": 172169486.0, |
| "reward": 0.3571484565734863, |
| "reward_std": 0.42836075127124784, |
| "rewards/reward_accuracy/mean": 0.2578125, |
| "rewards/reward_accuracy/std": 0.4280788689851761, |
| "rewards/reward_format/mean": 0.09933593943715095, |
| "rewards/reward_format/std": 0.005042777210474014, |
| "sampling/importance_sampling_ratio/max": 2.8507832288742065, |
| "sampling/importance_sampling_ratio/mean": 0.8760716199874878, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7661666393280029, |
| "sampling/sampling_logp_difference/mean": 0.008658151002600789, |
| "step": 1790, |
| "step_time": 20.84478442748077 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 2164.2, |
| "completions/max_terminated_length": 1813.3, |
| "completions/mean_length": 455.77578125, |
| "completions/mean_terminated_length": 435.5421447753906, |
| "completions/min_length": 102.0, |
| "completions/min_terminated_length": 102.0, |
| "entropy": 0.16285402891226114, |
| "epoch": 0.9977827050997783, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.291015625, |
| "learning_rate": 9.820100000000001e-06, |
| "loss": 0.0064, |
| "num_tokens": 173047879.0, |
| "reward": 0.4075781494379044, |
| "reward_std": 0.4529642194509506, |
| "rewards/reward_accuracy/mean": 0.30859375, |
| "rewards/reward_accuracy/std": 0.4522197723388672, |
| "rewards/reward_format/mean": 0.09898437783122063, |
| "rewards/reward_format/std": 0.007120213191956282, |
| "sampling/importance_sampling_ratio/max": 2.5714650392532348, |
| "sampling/importance_sampling_ratio/mean": 0.8651019632816315, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7081856369972229, |
| "sampling/sampling_logp_difference/mean": 0.008746904879808426, |
| "step": 1800, |
| "step_time": 20.43825912270695 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01015625, |
| "completions/max_length": 2772.3, |
| "completions/max_terminated_length": 1735.7, |
| "completions/mean_length": 495.740625, |
| "completions/mean_terminated_length": 469.45046081542966, |
| "completions/min_length": 103.1, |
| "completions/min_terminated_length": 103.1, |
| "entropy": 0.1630999685730785, |
| "epoch": 1.0033259423503327, |
| "frac_reward_zero_std": 0.6375, |
| "grad_norm": 0.41796875, |
| "learning_rate": 9.8191e-06, |
| "loss": 0.001, |
| "num_tokens": 173981987.0, |
| "reward": 0.3487500220537186, |
| "reward_std": 0.4256009340286255, |
| "rewards/reward_accuracy/mean": 0.25, |
| "rewards/reward_accuracy/std": 0.4248255014419556, |
| "rewards/reward_format/mean": 0.09875000268220901, |
| "rewards/reward_format/std": 0.009471583645790815, |
| "sampling/importance_sampling_ratio/max": 2.7684443473815916, |
| "sampling/importance_sampling_ratio/mean": 0.847916579246521, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7631311297416687, |
| "sampling/sampling_logp_difference/mean": 0.008696012198925018, |
| "step": 1810, |
| "step_time": 25.531781203206627 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00703125, |
| "completions/max_length": 2446.6, |
| "completions/max_terminated_length": 1870.8, |
| "completions/mean_length": 461.40390625, |
| "completions/mean_terminated_length": 443.2938262939453, |
| "completions/min_length": 73.6, |
| "completions/min_terminated_length": 73.6, |
| "entropy": 0.17612409894354641, |
| "epoch": 1.0088691796008868, |
| "frac_reward_zero_std": 0.63125, |
| "grad_norm": 0.294921875, |
| "learning_rate": 9.8181e-06, |
| "loss": -0.032, |
| "num_tokens": 174881304.0, |
| "reward": 0.40265626907348634, |
| "reward_std": 0.4481485038995743, |
| "rewards/reward_accuracy/mean": 0.30390625, |
| "rewards/reward_accuracy/std": 0.4472417920827866, |
| "rewards/reward_format/mean": 0.0987500011920929, |
| "rewards/reward_format/std": 0.009079474769532681, |
| "sampling/importance_sampling_ratio/max": 2.808813214302063, |
| "sampling/importance_sampling_ratio/mean": 0.8772522389888764, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6776934504508972, |
| "sampling/sampling_logp_difference/mean": 0.009098340384662151, |
| "step": 1820, |
| "step_time": 22.918096292112022 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2688.4, |
| "completions/max_terminated_length": 2023.5, |
| "completions/mean_length": 468.13046875, |
| "completions/mean_terminated_length": 443.4545166015625, |
| "completions/min_length": 78.6, |
| "completions/min_terminated_length": 78.6, |
| "entropy": 0.16901530819013716, |
| "epoch": 1.0144124168514412, |
| "frac_reward_zero_std": 0.575, |
| "grad_norm": 0.2421875, |
| "learning_rate": 9.817100000000001e-06, |
| "loss": -0.0193, |
| "num_tokens": 175774319.0, |
| "reward": 0.41812502443790434, |
| "reward_std": 0.46536935269832613, |
| "rewards/reward_accuracy/mean": 0.31953125, |
| "rewards/reward_accuracy/std": 0.4642303645610809, |
| "rewards/reward_format/mean": 0.09859375283122063, |
| "rewards/reward_format/std": 0.009591781487688422, |
| "sampling/importance_sampling_ratio/max": 2.7047069311141967, |
| "sampling/importance_sampling_ratio/mean": 0.8410809278488159, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6988677859306336, |
| "sampling/sampling_logp_difference/mean": 0.008781780349090695, |
| "step": 1830, |
| "step_time": 25.054711665492505 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00859375, |
| "completions/max_length": 2442.4, |
| "completions/max_terminated_length": 1674.5, |
| "completions/mean_length": 468.92734375, |
| "completions/mean_terminated_length": 446.1760284423828, |
| "completions/min_length": 77.1, |
| "completions/min_terminated_length": 77.1, |
| "entropy": 0.17745991088449956, |
| "epoch": 1.0199556541019956, |
| "frac_reward_zero_std": 0.6625, |
| "grad_norm": 0.111328125, |
| "learning_rate": 9.8161e-06, |
| "loss": -0.0112, |
| "num_tokens": 176679394.0, |
| "reward": 0.34722658395767214, |
| "reward_std": 0.4281252443790436, |
| "rewards/reward_accuracy/mean": 0.2484375, |
| "rewards/reward_accuracy/std": 0.4272913306951523, |
| "rewards/reward_format/mean": 0.09878906533122063, |
| "rewards/reward_format/std": 0.009462436847388744, |
| "sampling/importance_sampling_ratio/max": 2.7970136165618897, |
| "sampling/importance_sampling_ratio/mean": 0.8558044850826263, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7550158381462098, |
| "sampling/sampling_logp_difference/mean": 0.00920734005048871, |
| "step": 1840, |
| "step_time": 22.789986981917174 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2482.2, |
| "completions/max_terminated_length": 1834.6, |
| "completions/mean_length": 462.22578125, |
| "completions/mean_terminated_length": 450.05726928710936, |
| "completions/min_length": 76.8, |
| "completions/min_terminated_length": 76.8, |
| "entropy": 0.163102938933298, |
| "epoch": 1.0254988913525498, |
| "frac_reward_zero_std": 0.64375, |
| "grad_norm": 0.2021484375, |
| "learning_rate": 9.8151e-06, |
| "loss": -0.0111, |
| "num_tokens": 177565051.0, |
| "reward": 0.3655859664082527, |
| "reward_std": 0.43642138242721557, |
| "rewards/reward_accuracy/mean": 0.26640625, |
| "rewards/reward_accuracy/std": 0.43602022230625154, |
| "rewards/reward_format/mean": 0.09917969182133675, |
| "rewards/reward_format/std": 0.007080836221575737, |
| "sampling/importance_sampling_ratio/max": 2.6344032764434813, |
| "sampling/importance_sampling_ratio/mean": 0.8821995258331299, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7092655420303344, |
| "sampling/sampling_logp_difference/mean": 0.008752519730478525, |
| "step": 1850, |
| "step_time": 22.780140824709086 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2165.1, |
| "completions/max_terminated_length": 1854.2, |
| "completions/mean_length": 452.81953125, |
| "completions/mean_terminated_length": 436.44459228515626, |
| "completions/min_length": 69.4, |
| "completions/min_terminated_length": 69.4, |
| "entropy": 0.16716003119945527, |
| "epoch": 1.0310421286031042, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.22265625, |
| "learning_rate": 9.814100000000001e-06, |
| "loss": -0.0135, |
| "num_tokens": 178441340.0, |
| "reward": 0.3685156464576721, |
| "reward_std": 0.4384745329618454, |
| "rewards/reward_accuracy/mean": 0.26953125, |
| "rewards/reward_accuracy/std": 0.43781326711177826, |
| "rewards/reward_format/mean": 0.0989843763411045, |
| "rewards/reward_format/std": 0.007036413997411728, |
| "sampling/importance_sampling_ratio/max": 2.7918035984039307, |
| "sampling/importance_sampling_ratio/mean": 0.8618404388427734, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6218443036079406, |
| "sampling/sampling_logp_difference/mean": 0.008750134706497192, |
| "step": 1860, |
| "step_time": 20.221434327121823 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2227.0, |
| "completions/max_terminated_length": 1709.6, |
| "completions/mean_length": 440.14375, |
| "completions/mean_terminated_length": 423.5944458007813, |
| "completions/min_length": 90.5, |
| "completions/min_terminated_length": 90.5, |
| "entropy": 0.1613941782619804, |
| "epoch": 1.0365853658536586, |
| "frac_reward_zero_std": 0.60625, |
| "grad_norm": 0.1435546875, |
| "learning_rate": 9.813100000000001e-06, |
| "loss": -0.0033, |
| "num_tokens": 179295772.0, |
| "reward": 0.4387890875339508, |
| "reward_std": 0.46740240752696993, |
| "rewards/reward_accuracy/mean": 0.33984375, |
| "rewards/reward_accuracy/std": 0.4667419731616974, |
| "rewards/reward_format/mean": 0.09894531518220902, |
| "rewards/reward_format/std": 0.006066371779888868, |
| "sampling/importance_sampling_ratio/max": 2.7306801080703735, |
| "sampling/importance_sampling_ratio/mean": 0.8843936204910279, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7981494545936585, |
| "sampling/sampling_logp_difference/mean": 0.008437714260071515, |
| "step": 1870, |
| "step_time": 21.081364692142234 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.00390625, |
| "completions/max_length": 2284.4, |
| "completions/max_terminated_length": 1634.1, |
| "completions/mean_length": 454.85, |
| "completions/mean_terminated_length": 444.653564453125, |
| "completions/min_length": 82.4, |
| "completions/min_terminated_length": 82.4, |
| "entropy": 0.1661381094250828, |
| "epoch": 1.042128603104213, |
| "frac_reward_zero_std": 0.64375, |
| "grad_norm": 0.10986328125, |
| "learning_rate": 9.8121e-06, |
| "loss": 0.0063, |
| "num_tokens": 180183940.0, |
| "reward": 0.4079687774181366, |
| "reward_std": 0.4526926130056381, |
| "rewards/reward_accuracy/mean": 0.30859375, |
| "rewards/reward_accuracy/std": 0.45219989120960236, |
| "rewards/reward_format/mean": 0.09937500283122062, |
| "rewards/reward_format/std": 0.005684941355139017, |
| "sampling/importance_sampling_ratio/max": 2.763921093940735, |
| "sampling/importance_sampling_ratio/mean": 0.9332285225391388, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.7353523790836334, |
| "sampling/sampling_logp_difference/mean": 0.008719275146722794, |
| "step": 1880, |
| "step_time": 21.292445454606785 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2731.1, |
| "completions/max_terminated_length": 1931.4, |
| "completions/mean_length": 467.5140625, |
| "completions/mean_terminated_length": 442.7772644042969, |
| "completions/min_length": 97.2, |
| "completions/min_terminated_length": 97.2, |
| "entropy": 0.15958142513409257, |
| "epoch": 1.0476718403547671, |
| "frac_reward_zero_std": 0.625, |
| "grad_norm": 0.2314453125, |
| "learning_rate": 9.811100000000002e-06, |
| "loss": -0.0164, |
| "num_tokens": 181086310.0, |
| "reward": 0.3612500220537186, |
| "reward_std": 0.4373440146446228, |
| "rewards/reward_accuracy/mean": 0.2625, |
| "rewards/reward_accuracy/std": 0.4365715056657791, |
| "rewards/reward_format/mean": 0.09875000193715096, |
| "rewards/reward_format/std": 0.009169904608279466, |
| "sampling/importance_sampling_ratio/max": 2.7736751794815064, |
| "sampling/importance_sampling_ratio/mean": 0.880827909708023, |
| "sampling/importance_sampling_ratio/min": 0.0169302299618721, |
| "sampling/sampling_logp_difference/max": 0.7095172345638275, |
| "sampling/sampling_logp_difference/mean": 0.008578233513981104, |
| "step": 1890, |
| "step_time": 25.76963075506501 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2279.1, |
| "completions/max_terminated_length": 1877.9, |
| "completions/mean_length": 451.80234375, |
| "completions/mean_terminated_length": 439.56719360351565, |
| "completions/min_length": 88.4, |
| "completions/min_terminated_length": 88.4, |
| "entropy": 0.16308262925595046, |
| "epoch": 1.0532150776053215, |
| "frac_reward_zero_std": 0.70625, |
| "grad_norm": 0.232421875, |
| "learning_rate": 9.810100000000001e-06, |
| "loss": 0.0122, |
| "num_tokens": 181971137.0, |
| "reward": 0.3696484625339508, |
| "reward_std": 0.4406398504972458, |
| "rewards/reward_accuracy/mean": 0.2703125, |
| "rewards/reward_accuracy/std": 0.44015699326992036, |
| "rewards/reward_format/mean": 0.09933594092726708, |
| "rewards/reward_format/std": 0.0066198139451444146, |
| "sampling/importance_sampling_ratio/max": 2.7761741638183595, |
| "sampling/importance_sampling_ratio/mean": 0.8833258211612701, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6468923389911652, |
| "sampling/sampling_logp_difference/mean": 0.008550268411636353, |
| "step": 1900, |
| "step_time": 21.016413699043916 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.3978811032066006e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.3978811032066006e-05, |
| "completions/clipped_ratio": 0.009375, |
| "completions/max_length": 2168.6, |
| "completions/max_terminated_length": 1720.7, |
| "completions/mean_length": 479.6109375, |
| "completions/mean_terminated_length": 455.1732482910156, |
| "completions/min_length": 124.3, |
| "completions/min_terminated_length": 124.3, |
| "entropy": 0.1525016195140779, |
| "epoch": 0.529379157427938, |
| "frac_reward_zero_std": 0.65, |
| "grad_norm": 0.35546875, |
| "learning_rate": 9.9991e-06, |
| "loss": -0.0011, |
| "num_tokens": 182423000.0, |
| "reward": 0.35523439198732376, |
| "reward_std": 0.3971613973379135, |
| "rewards/reward_accuracy/mean": 0.25625, |
| "rewards/reward_accuracy/std": 0.39657233357429506, |
| "rewards/reward_format/mean": 0.0989843763411045, |
| "rewards/reward_format/std": 0.005228776205331087, |
| "sampling/importance_sampling_ratio/max": 2.4895227909088136, |
| "sampling/importance_sampling_ratio/mean": 0.8932334423065186, |
| "sampling/importance_sampling_ratio/min": 0.014378012344241142, |
| "sampling/sampling_logp_difference/max": 0.5629366040229797, |
| "sampling/sampling_logp_difference/mean": 0.007852759258821607, |
| "step": 1910, |
| "step_time": 19.71194686428644 |
| }, |
| { |
| "clip_ratio/high_max": 8.615316328359768e-05, |
| "clip_ratio/high_mean": 2.153829082089942e-05, |
| "clip_ratio/low_mean": 2.1334748271328863e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.287303909222828e-05, |
| "completions/clipped_ratio": 0.0171875, |
| "completions/max_length": 2737.4, |
| "completions/max_terminated_length": 1335.0, |
| "completions/mean_length": 459.9109375, |
| "completions/mean_terminated_length": 414.2674591064453, |
| "completions/min_length": 87.6, |
| "completions/min_terminated_length": 87.6, |
| "entropy": 0.14635315956547856, |
| "epoch": 0.532150776053215, |
| "frac_reward_zero_std": 0.7125, |
| "grad_norm": 0.28125, |
| "learning_rate": 9.9981e-06, |
| "loss": -0.004, |
| "num_tokens": 182863887.0, |
| "reward": 0.41359376907348633, |
| "reward_std": 0.45846299827098846, |
| "rewards/reward_accuracy/mean": 0.315625, |
| "rewards/reward_accuracy/std": 0.45685032308101653, |
| "rewards/reward_format/mean": 0.09796875119209289, |
| "rewards/reward_format/std": 0.012197112571448088, |
| "sampling/importance_sampling_ratio/max": 2.4620474576950073, |
| "sampling/importance_sampling_ratio/mean": 0.8722427129745484, |
| "sampling/importance_sampling_ratio/min": 0.04644599184393883, |
| "sampling/sampling_logp_difference/max": 0.6076258182525635, |
| "sampling/sampling_logp_difference/mean": 0.0073147373739629986, |
| "step": 1920, |
| "step_time": 24.950853064330296 |
| }, |
| { |
| "clip_ratio/high_max": 8.87064656126313e-05, |
| "clip_ratio/high_mean": 2.2176616403157824e-05, |
| "clip_ratio/low_mean": 1.7776495224097743e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.9953111627255564e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1501.2, |
| "completions/max_terminated_length": 1501.2, |
| "completions/mean_length": 487.55625, |
| "completions/mean_terminated_length": 487.55625, |
| "completions/min_length": 112.7, |
| "completions/min_terminated_length": 112.7, |
| "entropy": 0.16149511388503016, |
| "epoch": 0.5349223946784922, |
| "frac_reward_zero_std": 0.5875, |
| "grad_norm": 0.1337890625, |
| "learning_rate": 9.997100000000001e-06, |
| "loss": 0.024, |
| "num_tokens": 183324419.0, |
| "reward": 0.3289843946695328, |
| "reward_std": 0.3837434411048889, |
| "rewards/reward_accuracy/mean": 0.2296875, |
| "rewards/reward_accuracy/std": 0.38334057927131654, |
| "rewards/reward_format/mean": 0.09929687753319741, |
| "rewards/reward_format/std": 0.004442050866782665, |
| "sampling/importance_sampling_ratio/max": 2.614134967327118, |
| "sampling/importance_sampling_ratio/mean": 0.8654231250286102, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.6586638987064362, |
| "sampling/sampling_logp_difference/mean": 0.008197818091139198, |
| "step": 1930, |
| "step_time": 14.022369165811687 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015229038544930517, |
| "clip_ratio/high_mean": 3.807259636232629e-05, |
| "clip_ratio/low_mean": 1.3880951155442744e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 5.195354751776904e-05, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 1925.0, |
| "completions/max_terminated_length": 1421.8, |
| "completions/mean_length": 502.4953125, |
| "completions/mean_terminated_length": 482.3774139404297, |
| "completions/min_length": 142.1, |
| "completions/min_terminated_length": 142.1, |
| "entropy": 0.14076343039050698, |
| "epoch": 0.5376940133037694, |
| "frac_reward_zero_std": 0.6125, |
| "grad_norm": 0.158203125, |
| "learning_rate": 9.9961e-06, |
| "loss": 0.0011, |
| "num_tokens": 183794288.0, |
| "reward": 0.38968752324581146, |
| "reward_std": 0.45384337604045866, |
| "rewards/reward_accuracy/mean": 0.290625, |
| "rewards/reward_accuracy/std": 0.45310727059841155, |
| "rewards/reward_format/mean": 0.09906250238418579, |
| "rewards/reward_format/std": 0.00532838637009263, |
| "sampling/importance_sampling_ratio/max": 2.5872971534729006, |
| "sampling/importance_sampling_ratio/mean": 0.8563449144363403, |
| "sampling/importance_sampling_ratio/min": 0.031803447753190994, |
| "sampling/sampling_logp_difference/max": 0.5164010047912597, |
| "sampling/sampling_logp_difference/mean": 0.00755718257278204, |
| "step": 1940, |
| "step_time": 18.19395874668844 |
| }, |
| { |
| "clip_ratio/high_max": 7.734961109235882e-05, |
| "clip_ratio/high_mean": 1.9337402773089706e-05, |
| "clip_ratio/low_mean": 2.1359210131777216e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.0696612904866926e-05, |
| "completions/clipped_ratio": 0.0015625, |
| "completions/max_length": 1748.2, |
| "completions/max_terminated_length": 1581.1, |
| "completions/mean_length": 460.7140625, |
| "completions/mean_terminated_length": 456.6415710449219, |
| "completions/min_length": 99.0, |
| "completions/min_terminated_length": 99.0, |
| "entropy": 0.1522112991195172, |
| "epoch": 0.5404656319290465, |
| "frac_reward_zero_std": 0.55, |
| "grad_norm": 0.3828125, |
| "learning_rate": 9.9951e-06, |
| "loss": 0.0048, |
| "num_tokens": 184243529.0, |
| "reward": 0.3388281434774399, |
| "reward_std": 0.3998119443655014, |
| "rewards/reward_accuracy/mean": 0.2390625, |
| "rewards/reward_accuracy/std": 0.3998646795749664, |
| "rewards/reward_format/mean": 0.0997656263411045, |
| "rewards/reward_format/std": 0.0013886408880352974, |
| "sampling/importance_sampling_ratio/max": 2.7144407510757445, |
| "sampling/importance_sampling_ratio/mean": 0.9173040866851807, |
| "sampling/importance_sampling_ratio/min": 0.00316086420789361, |
| "sampling/sampling_logp_difference/max": 0.6160829901695252, |
| "sampling/sampling_logp_difference/mean": 0.007597061572596431, |
| "step": 1950, |
| "step_time": 16.089866680046544 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.548155380281969e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 2.548155380281969e-05, |
| "completions/clipped_ratio": 0.0046875, |
| "completions/max_length": 2086.5, |
| "completions/max_terminated_length": 1612.1, |
| "completions/mean_length": 471.91875, |
| "completions/mean_terminated_length": 459.7032806396484, |
| "completions/min_length": 108.3, |
| "completions/min_terminated_length": 108.3, |
| "entropy": 0.16093067219480872, |
| "epoch": 0.5432372505543237, |
| "frac_reward_zero_std": 0.7125, |
| "grad_norm": 0.171875, |
| "learning_rate": 9.994100000000001e-06, |
| "loss": -0.018, |
| "num_tokens": 184689333.0, |
| "reward": 0.2525000162422657, |
| "reward_std": 0.31939140260219573, |
| "rewards/reward_accuracy/mean": 0.153125, |
| "rewards/reward_accuracy/std": 0.31896214932203293, |
| "rewards/reward_format/mean": 0.09937500134110451, |
| "rewards/reward_format/std": 0.004253681190311909, |
| "sampling/importance_sampling_ratio/max": 2.5558556795120237, |
| "sampling/importance_sampling_ratio/mean": 0.856280654668808, |
| "sampling/importance_sampling_ratio/min": 0.018020515330135822, |
| "sampling/sampling_logp_difference/max": 0.6166099369525909, |
| "sampling/sampling_logp_difference/mean": 0.008554295590147375, |
| "step": 1960, |
| "step_time": 19.019149316940457 |
| }, |
| { |
| "clip_ratio/high_max": 5.326704704202712e-05, |
| "clip_ratio/high_mean": 1.331676176050678e-05, |
| "clip_ratio/low_mean": 1.6740533192205474e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.0057294952712255e-05, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 2283.1, |
| "completions/max_terminated_length": 1683.4, |
| "completions/mean_length": 431.39375, |
| "completions/mean_terminated_length": 414.7422210693359, |
| "completions/min_length": 102.6, |
| "completions/min_terminated_length": 102.6, |
| "entropy": 0.16841390491463243, |
| "epoch": 0.5460088691796009, |
| "frac_reward_zero_std": 0.65, |
| "grad_norm": 0.1630859375, |
| "learning_rate": 9.993100000000001e-06, |
| "loss": -0.0035, |
| "num_tokens": 185114449.0, |
| "reward": 0.41164064556360247, |
| "reward_std": 0.4358812004327774, |
| "rewards/reward_accuracy/mean": 0.3125, |
| "rewards/reward_accuracy/std": 0.4350589022040367, |
| "rewards/reward_format/mean": 0.09914062619209289, |
| "rewards/reward_format/std": 0.005139226280152798, |
| "sampling/importance_sampling_ratio/max": 2.606655788421631, |
| "sampling/importance_sampling_ratio/mean": 0.8908221423625946, |
| "sampling/importance_sampling_ratio/min": 0.011747117713093757, |
| "sampling/sampling_logp_difference/max": 0.5437958002090454, |
| "sampling/sampling_logp_difference/mean": 0.008746534725651145, |
| "step": 1970, |
| "step_time": 20.9888335427735 |
| }, |
| { |
| "clip_ratio/high_max": 0.00011373838497092947, |
| "clip_ratio/high_mean": 2.8434596242732368e-05, |
| "clip_ratio/low_mean": 1.2412775458869873e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.084737170160224e-05, |
| "completions/clipped_ratio": 0.00625, |
| "completions/max_length": 1836.0, |
| "completions/max_terminated_length": 1270.5, |
| "completions/mean_length": 454.5640625, |
| "completions/mean_terminated_length": 438.47943420410155, |
| "completions/min_length": 145.7, |
| "completions/min_terminated_length": 145.7, |
| "entropy": 0.14269639370031656, |
| "epoch": 0.5487804878048781, |
| "frac_reward_zero_std": 0.5375, |
| "grad_norm": 0.2099609375, |
| "learning_rate": 9.9921e-06, |
| "loss": 0.0351, |
| "num_tokens": 185559050.0, |
| "reward": 0.40375002324581144, |
| "reward_std": 0.44142946898937224, |
| "rewards/reward_accuracy/mean": 0.3046875, |
| "rewards/reward_accuracy/std": 0.44089526534080503, |
| "rewards/reward_format/mean": 0.09906250163912773, |
| "rewards/reward_format/std": 0.005685210693627596, |
| "sampling/importance_sampling_ratio/max": 2.739349102973938, |
| "sampling/importance_sampling_ratio/mean": 0.8985099852085113, |
| "sampling/importance_sampling_ratio/min": 0.023648651689291, |
| "sampling/sampling_logp_difference/max": 0.6167889952659606, |
| "sampling/sampling_logp_difference/mean": 0.007511630421504378, |
| "step": 1980, |
| "step_time": 17.166691356664522 |
| }, |
| { |
| "clip_ratio/high_max": 8.053907949943096e-05, |
| "clip_ratio/high_mean": 2.013476987485774e-05, |
| "clip_ratio/low_mean": 4.855445013163262e-05, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.868922000649035e-05, |
| "completions/clipped_ratio": 0.0125, |
| "completions/max_length": 2148.2, |
| "completions/max_terminated_length": 1599.2, |
| "completions/mean_length": 474.5484375, |
| "completions/mean_terminated_length": 442.478759765625, |
| "completions/min_length": 96.5, |
| "completions/min_terminated_length": 96.5, |
| "entropy": 0.1565450391266495, |
| "epoch": 0.5515521064301552, |
| "frac_reward_zero_std": 0.5125, |
| "grad_norm": 0.26171875, |
| "learning_rate": 9.991100000000002e-06, |
| "loss": 0.0128, |
| "num_tokens": 186015833.0, |
| "reward": 0.40000001788139344, |
| "reward_std": 0.4409543454647064, |
| "rewards/reward_accuracy/mean": 0.3015625, |
| "rewards/reward_accuracy/std": 0.4399972319602966, |
| "rewards/reward_format/mean": 0.09843750074505805, |
| "rewards/reward_format/std": 0.008641463797539472, |
| "sampling/importance_sampling_ratio/max": 2.64941463470459, |
| "sampling/importance_sampling_ratio/mean": 0.8825161218643188, |
| "sampling/importance_sampling_ratio/min": 0.01930158957839012, |
| "sampling/sampling_logp_difference/max": 0.5945345640182496, |
| "sampling/sampling_logp_difference/mean": 0.00838247281499207, |
| "step": 1990, |
| "step_time": 20.158592392504215 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015346245927503333, |
| "clip_ratio/high_mean": 3.836561481875833e-05, |
| "clip_ratio/low_mean": 9.1552734375e-06, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.7520888256258334e-05, |
| "completions/clipped_ratio": 0.0140625, |
| "completions/max_length": 2416.5, |
| "completions/max_terminated_length": 1447.4, |
| "completions/mean_length": 452.75, |
| "completions/mean_terminated_length": 415.02337036132815, |
| "completions/min_length": 98.7, |
| "completions/min_terminated_length": 98.7, |
| "entropy": 0.14157701721414923, |
| "epoch": 0.5543237250554324, |
| "frac_reward_zero_std": 0.6125, |
| "grad_norm": 0.4921875, |
| "learning_rate": 9.990100000000001e-06, |
| "loss": -0.0112, |
| "num_tokens": 186454601.0, |
| "reward": 0.5297656506299973, |
| "reward_std": 0.4936007231473923, |
| "rewards/reward_accuracy/mean": 0.43125, |
| "rewards/reward_accuracy/std": 0.4922992080450058, |
| "rewards/reward_format/mean": 0.09851562604308128, |
| "rewards/reward_format/std": 0.00856757303699851, |
| "sampling/importance_sampling_ratio/max": 2.4509358406066895, |
| "sampling/importance_sampling_ratio/mean": 0.8909220278263092, |
| "sampling/importance_sampling_ratio/min": 0.0, |
| "sampling/sampling_logp_difference/max": 0.5613163709640503, |
| "sampling/sampling_logp_difference/mean": 0.007465607533231377, |
| "step": 2000, |
| "step_time": 22.591339117567987 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 100000, |
| "num_input_tokens_seen": 186454601, |
| "num_train_epochs": 28, |
| "save_steps": 100, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|