{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 500, "global_step": 384, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 23.625, "completions/mean_terminated_length": 23.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09965698793530464, "epoch": 0.0078125, "frac_reward_zero_std": 0.5, "grad_norm": 1.13743257522583, "kl": 9.029518110992285e-08, "learning_rate": 0.0, "loss": -0.06901008635759354, "num_tokens": 4097.0, "reward": 0.768750011920929, "reward_std": 0.4300643801689148, "rewards/reward_fn/mean": 0.768750011920929, "rewards/reward_fn/std": 0.4300643801689148, "step": 1, "step_time": 7.910418492999952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13235441222786903, "epoch": 0.015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 1.282051282051282e-07, "loss": 0.0, "num_tokens": 9674.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 2, "step_time": 7.016720726000017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 22.5, "completions/mean_terminated_length": 22.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11623429134488106, "epoch": 0.0234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0009358773822896183, "kl": 8.32239061310247e-06, "learning_rate": 2.564102564102564e-07, "loss": 8.256898098579768e-08, "num_tokens": 14558.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 3, "step_time": 6.771791392000068 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 23.5, "completions/mean_terminated_length": 23.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09514202550053596, "epoch": 0.03125, "frac_reward_zero_std": 0.5, "grad_norm": 1.5667312145233154, "kl": 3.819915718850098e-06, "learning_rate": 3.846153846153847e-07, "loss": -0.08508844673633575, "num_tokens": 18626.0, "reward": 0.8812500238418579, "reward_std": 0.3358757197856903, "rewards/reward_fn/mean": 0.8812500238418579, "rewards/reward_fn/std": 0.3358757197856903, "step": 4, "step_time": 5.745465063999859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.14030911028385162, "epoch": 0.0390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0005072976928204298, "kl": 4.893604227618198e-06, "learning_rate": 5.128205128205128e-07, "loss": 4.175808498985134e-08, "num_tokens": 24180.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 5, "step_time": 7.145598770999982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.375, "completions/mean_terminated_length": 21.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11003290489315987, "epoch": 0.046875, "frac_reward_zero_std": 0.5, "grad_norm": 3.21610426902771, "kl": 1.439371067135653e-05, "learning_rate": 6.41025641025641e-07, "loss": 0.008770093321800232, "num_tokens": 28935.0, "reward": 0.887499988079071, "reward_std": 0.3181980550289154, "rewards/reward_fn/mean": 0.887499988079071, "rewards/reward_fn/std": 0.3181980550289154, "step": 6, "step_time": 6.531211202000009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.625, "completions/mean_terminated_length": 21.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12766291946172714, "epoch": 0.0546875, "frac_reward_zero_std": 0.0, "grad_norm": 2.585819959640503, "kl": 2.266431465614005e-06, "learning_rate": 7.692307692307694e-07, "loss": -0.25185173749923706, "num_tokens": 32988.0, "reward": 0.643750011920929, "reward_std": 0.49384605884552, "rewards/reward_fn/mean": 0.643750011920929, "rewards/reward_fn/std": 0.49384605884552, "step": 7, "step_time": 5.520926363000058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.75, "completions/mean_terminated_length": 21.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11385262385010719, "epoch": 0.0625, "frac_reward_zero_std": 1.0, "grad_norm": 0.00018979530432261527, "kl": 3.1351814868685324e-06, "learning_rate": 8.974358974358975e-07, "loss": 3.0823137819879776e-08, "num_tokens": 37866.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 8, "step_time": 6.259193151999966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1592414677143097, "epoch": 0.0703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0010657254606485367, "kl": 1.2110989018765395e-05, "learning_rate": 1.0256410256410257e-06, "loss": 1.211098918929565e-07, "num_tokens": 43344.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 9, "step_time": 4.96420772700003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1171259917318821, "epoch": 0.078125, "frac_reward_zero_std": 0.5, "grad_norm": 1.561084508895874, "kl": 7.743469495835598e-06, "learning_rate": 1.153846153846154e-06, "loss": -0.05262097343802452, "num_tokens": 47400.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 10, "step_time": 5.405768857999988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 13.0, "completions/max_terminated_length": 13.0, "completions/mean_length": 13.0, "completions/mean_terminated_length": 13.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.16070035845041275, "epoch": 0.0859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.00039809694862924516, "kl": 4.659478577195841e-06, "learning_rate": 1.282051282051282e-06, "loss": 4.659478491930713e-08, "num_tokens": 52904.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 11, "step_time": 5.233364768000001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.25, "completions/mean_terminated_length": 23.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11492524296045303, "epoch": 0.09375, "frac_reward_zero_std": 0.0, "grad_norm": NaN, "kl": 8.350619191332953e-05, "learning_rate": 1.4102564102564104e-06, "loss": -0.1988813430070877, "num_tokens": 57022.0, "reward": 0.7687499523162842, "reward_std": 0.4284002482891083, "rewards/reward_fn/mean": 0.7687499523162842, "rewards/reward_fn/std": 0.4284002482891083, "step": 12, "step_time": 5.546596589000046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.17772966623306274, "epoch": 0.1015625, "frac_reward_zero_std": 0.5, "grad_norm": 1.7892787456512451, "kl": 1.1041468042094493e-05, "learning_rate": 1.5384615384615387e-06, "loss": -0.147029310464859, "num_tokens": 61727.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 13, "step_time": 6.295519733999981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13657044991850853, "epoch": 0.109375, "frac_reward_zero_std": 0.5, "grad_norm": 2.4697511196136475, "kl": 0.00011959002404182684, "learning_rate": 1.6666666666666667e-06, "loss": -0.07215305417776108, "num_tokens": 65771.0, "reward": 0.7875000238418579, "reward_std": 0.3934735357761383, "rewards/reward_fn/mean": 0.7875000238418579, "rewards/reward_fn/std": 0.3934735655784607, "step": 14, "step_time": 5.657076707999977 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10727399960160255, "epoch": 0.1171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0006498436559922993, "kl": 2.7017442334908992e-05, "learning_rate": 1.794871794871795e-06, "loss": 2.6791002483150805e-07, "num_tokens": 70595.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 15, "step_time": 9.642551118000029 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 22.0, "completions/mean_terminated_length": 22.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09909634664654732, "epoch": 0.125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0024827050510793924, "kl": 6.381553976098076e-05, "learning_rate": 1.9230769230769234e-06, "loss": 6.585330538655398e-07, "num_tokens": 75331.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 16, "step_time": 7.090756369000019 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11970487236976624, "epoch": 0.1328125, "frac_reward_zero_std": 0.0, "grad_norm": 2.5904784202575684, "kl": 0.0007240207705763169, "learning_rate": 2.0512820512820513e-06, "loss": -0.2588059902191162, "num_tokens": 79512.0, "reward": 0.5437500476837158, "reward_std": 0.49021676182746887, "rewards/reward_fn/mean": 0.5437500476837158, "rewards/reward_fn/std": 0.49021679162979126, "step": 17, "step_time": 5.971762764999994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10251206904649734, "epoch": 0.140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.003479904029518366, "kl": 0.00019486098608467728, "learning_rate": 2.1794871794871797e-06, "loss": 1.9753097149077803e-06, "num_tokens": 84289.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 18, "step_time": 6.534459332000097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 16.875, "completions/mean_terminated_length": 16.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13044045120477676, "epoch": 0.1484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.006375588476657867, "kl": 0.00021761858806712553, "learning_rate": 2.307692307692308e-06, "loss": 2.2181316126079764e-06, "num_tokens": 89848.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 19, "step_time": 8.074243104000061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07144641503691673, "epoch": 0.15625, "frac_reward_zero_std": 0.5, "grad_norm": 1.383427381515503, "kl": 0.00042095681419596076, "learning_rate": 2.435897435897436e-06, "loss": -0.111912801861763, "num_tokens": 93977.0, "reward": 0.8812500238418579, "reward_std": 0.3358757197856903, "rewards/reward_fn/mean": 0.8812500238418579, "rewards/reward_fn/std": 0.3358757197856903, "step": 20, "step_time": 5.788662745999886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10241465270519257, "epoch": 0.1640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0049729375168681145, "kl": 0.00047132828331086785, "learning_rate": 2.564102564102564e-06, "loss": 4.8525371312280186e-06, "num_tokens": 98852.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 21, "step_time": 6.948417052999957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.75, "completions/mean_terminated_length": 24.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09643011912703514, "epoch": 0.171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.006859563756734133, "kl": 0.0007727623451501131, "learning_rate": 2.6923076923076923e-06, "loss": 6.733347163390135e-06, "num_tokens": 103734.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 22, "step_time": 7.359714186000019 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 23.375, "completions/mean_terminated_length": 23.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.42096097022295, "epoch": 0.1796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.005077087786048651, "kl": 0.0007580214296467602, "learning_rate": 2.8205128205128207e-06, "loss": 7.698860827076714e-06, "num_tokens": 108625.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 23, "step_time": 8.31426027000009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 30.0, "completions/mean_terminated_length": 30.0, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.05439547263085842, "epoch": 0.1875, "frac_reward_zero_std": 1.0, "grad_norm": 0.000995291629806161, "kl": 0.0004853812279179692, "learning_rate": 2.948717948717949e-06, "loss": 4.853812242799904e-06, "num_tokens": 112781.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 24, "step_time": 5.9903554890000805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 24.375, "completions/mean_terminated_length": 24.375, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.10691225156188011, "epoch": 0.1953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.011116056703031063, "kl": 0.0012052947713527828, "learning_rate": 3.0769230769230774e-06, "loss": 1.1946971426368691e-05, "num_tokens": 117612.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 25, "step_time": 7.288805328999956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1299378089606762, "epoch": 0.203125, "frac_reward_zero_std": 0.5, "grad_norm": 1.851197361946106, "kl": 0.002050661132670939, "learning_rate": 3.205128205128206e-06, "loss": -0.11245696991682053, "num_tokens": 122448.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 26, "step_time": 7.037999247000016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.375, "completions/mean_terminated_length": 21.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07765422388911247, "epoch": 0.2109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.00900979246944189, "kl": 0.001701147179119289, "learning_rate": 3.3333333333333333e-06, "loss": 1.7008303984766826e-05, "num_tokens": 127211.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 27, "step_time": 6.768718518000014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.053295012563467026, "epoch": 0.21875, "frac_reward_zero_std": 0.5, "grad_norm": 1.0311310291290283, "kl": 0.005389484780607745, "learning_rate": 3.4615384615384617e-06, "loss": -0.10706700384616852, "num_tokens": 131351.0, "reward": 0.893750011920929, "reward_std": 0.3005203604698181, "rewards/reward_fn/mean": 0.893750011920929, "rewards/reward_fn/std": 0.3005203604698181, "step": 28, "step_time": 5.859235952000063 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 18.5, "completions/mean_terminated_length": 18.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08729634806513786, "epoch": 0.2265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.11862029135227203, "kl": 0.00995962810702622, "learning_rate": 3.58974358974359e-06, "loss": 0.00010122068488271907, "num_tokens": 136067.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 29, "step_time": 7.392926845999909 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.049397675320506096, "epoch": 0.234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.013916196301579475, "kl": 0.004311037337174639, "learning_rate": 3.7179487179487184e-06, "loss": 3.796327655436471e-05, "num_tokens": 140216.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 30, "step_time": 5.707905451999977 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 44.625, "completions/mean_terminated_length": 32.71428680419922, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.3870660662651062, "epoch": 0.2421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.00857287272810936, "kl": 0.0027850136975757778, "learning_rate": 3.846153846153847e-06, "loss": 2.3635355319129303e-05, "num_tokens": 145949.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 31, "step_time": 14.826277505999997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 29.25, "completions/mean_terminated_length": 29.25, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.15847552567720413, "epoch": 0.25, "frac_reward_zero_std": 0.5, "grad_norm": 3.0680534839630127, "kl": 0.005178321152925491, "learning_rate": 3.974358974358974e-06, "loss": 0.021413102746009827, "num_tokens": 150875.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 32, "step_time": 7.193657256999927 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.375, "completions/mean_terminated_length": 27.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.036486370489001274, "epoch": 0.2578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.013429846614599228, "kl": 0.0038138862000778317, "learning_rate": 4.102564102564103e-06, "loss": 3.6106022889725864e-05, "num_tokens": 155090.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 33, "step_time": 5.989709379999908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.75, "completions/mean_terminated_length": 24.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06528343632817268, "epoch": 0.265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.23420189321041107, "kl": 0.015011879149824381, "learning_rate": 4.230769230769231e-06, "loss": 0.00015618561883457005, "num_tokens": 159860.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 34, "step_time": 7.343351643999995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.75, "completions/mean_terminated_length": 24.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09101804345846176, "epoch": 0.2734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0637916773557663, "kl": 0.011130816768854856, "learning_rate": 4.358974358974359e-06, "loss": 0.00011102524877060205, "num_tokens": 165430.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 35, "step_time": 7.710919977999993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.06510375998914242, "epoch": 0.28125, "frac_reward_zero_std": 1.0, "grad_norm": 0.03826780989766121, "kl": 0.0053558857180178165, "learning_rate": 4.487179487179488e-06, "loss": 5.355885878088884e-05, "num_tokens": 170254.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 36, "step_time": 7.2804435799999965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.125, "completions/mean_terminated_length": 27.125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.043790558353066444, "epoch": 0.2890625, "frac_reward_zero_std": 0.5, "grad_norm": 1.402679681777954, "kl": 0.02490220731124282, "learning_rate": 4.615384615384616e-06, "loss": 0.03481510281562805, "num_tokens": 174395.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 37, "step_time": 5.756117628000084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 18.875, "completions/mean_terminated_length": 18.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12046026065945625, "epoch": 0.296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.054251085966825485, "kl": 0.018674671184271574, "learning_rate": 4.743589743589744e-06, "loss": 0.00016035939916037023, "num_tokens": 179942.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 38, "step_time": 7.3252331490000415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 26.375, "completions/mean_terminated_length": 26.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05612089857459068, "epoch": 0.3046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.039188120514154434, "kl": 0.012932237819768488, "learning_rate": 4.871794871794872e-06, "loss": 0.00010084287350764498, "num_tokens": 184741.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 39, "step_time": 7.515498236999974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.75, "completions/mean_terminated_length": 24.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09634075127542019, "epoch": 0.3125, "frac_reward_zero_std": 1.0, "grad_norm": 0.018983978778123856, "kl": 0.005221477011218667, "learning_rate": 5e-06, "loss": 5.3893018048256636e-05, "num_tokens": 189627.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 40, "step_time": 7.176626092999982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.25, "completions/mean_terminated_length": 21.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06109755299985409, "epoch": 0.3203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.030976427718997, "kl": 0.009227571543306112, "learning_rate": 4.999896350176413e-06, "loss": 9.208174014929682e-05, "num_tokens": 194497.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 41, "step_time": 6.542497393999952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 20.625, "completions/mean_terminated_length": 20.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.17987589165568352, "epoch": 0.328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0431177020072937, "kl": 0.015479911118745804, "learning_rate": 4.999585409300281e-06, "loss": 0.0001403249625582248, "num_tokens": 200082.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 42, "step_time": 7.701265004999868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07510707713663578, "epoch": 0.3359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.03998936340212822, "kl": 0.012762806843966246, "learning_rate": 4.999067203154777e-06, "loss": 0.00010049781849374995, "num_tokens": 204927.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 43, "step_time": 7.225048154999968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06112176366150379, "epoch": 0.34375, "frac_reward_zero_std": 1.0, "grad_norm": 0.028579436242580414, "kl": 0.01107706013135612, "learning_rate": 4.998341774709482e-06, "loss": 0.00010365620255470276, "num_tokens": 209771.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 44, "step_time": 6.643636920000063 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 27.875, "completions/mean_terminated_length": 27.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12446310371160507, "epoch": 0.3515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.012319399043917656, "kl": 0.004381780279800296, "learning_rate": 4.9974091841168195e-06, "loss": 4.3796011595986784e-05, "num_tokens": 214682.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 45, "step_time": 7.909422166999889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 22.375, "completions/mean_terminated_length": 22.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10363675281405449, "epoch": 0.359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.04150591790676117, "kl": 0.007062526885420084, "learning_rate": 4.99626950870707e-06, "loss": 6.894840043969452e-05, "num_tokens": 220237.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 46, "step_time": 7.691492860000039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.030814163386821747, "epoch": 0.3671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.013106024824082851, "kl": 0.0039410877507179976, "learning_rate": 4.994922842981958e-06, "loss": 3.779985854635015e-05, "num_tokens": 224417.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 47, "step_time": 5.999039098000026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.625, "completions/mean_terminated_length": 16.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1171766109764576, "epoch": 0.375, "frac_reward_zero_std": 1.0, "grad_norm": 0.04344555363059044, "kl": 0.010649031959474087, "learning_rate": 4.993369298606817e-06, "loss": 0.00010028588440036401, "num_tokens": 229926.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 48, "step_time": 7.581720861000008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.625, "completions/mean_terminated_length": 13.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1510508954524994, "epoch": 0.3828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.04870390519499779, "kl": 0.011529957875609398, "learning_rate": 4.991609004401324e-06, "loss": 0.0001149907911894843, "num_tokens": 235435.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 49, "step_time": 5.576136509999969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.0, "completions/mean_terminated_length": 17.0, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.13440731912851334, "epoch": 0.390625, "frac_reward_zero_std": 0.5, "grad_norm": 1.8115813732147217, "kl": 0.20483593340031803, "learning_rate": 4.989642106328829e-06, "loss": -0.1266314685344696, "num_tokens": 240183.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 50, "step_time": 6.442843831000005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.375, "completions/mean_terminated_length": 27.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.029575850814580917, "epoch": 0.3984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.006918889936059713, "kl": 0.0022675381042063236, "learning_rate": 4.98746876748424e-06, "loss": 2.1996882423991337e-05, "num_tokens": 244150.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 51, "step_time": 5.648842897999998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.75, "completions/mean_terminated_length": 13.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12618596106767654, "epoch": 0.40625, "frac_reward_zero_std": 1.0, "grad_norm": 0.04183642938733101, "kl": 0.007541614351794124, "learning_rate": 4.985089168080509e-06, "loss": 7.499127241317183e-05, "num_tokens": 249636.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 52, "step_time": 5.985882786999923 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.027834360487759113, "epoch": 0.4140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.005657592322677374, "kl": 0.0021601368789561093, "learning_rate": 4.982503505433683e-06, "loss": 2.106852480210364e-05, "num_tokens": 253604.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 53, "step_time": 5.815157560000102 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 30.0, "completions/mean_terminated_length": 30.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.049247266724705696, "epoch": 0.421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.008218275383114815, "kl": 0.0025316422106698155, "learning_rate": 4.979711993946543e-06, "loss": 2.4405037038377486e-05, "num_tokens": 257780.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 54, "step_time": 7.388005351000061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 38.25, "completions/mean_terminated_length": 25.428571701049805, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.14161667972803116, "epoch": 0.4296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.011416819877922535, "kl": 0.0019309421186335385, "learning_rate": 4.976714865090827e-06, "loss": 2.0552859496092424e-05, "num_tokens": 263510.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 55, "step_time": 15.045006353999952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11700041592121124, "epoch": 0.4375, "frac_reward_zero_std": 1.0, "grad_norm": 0.01164371706545353, "kl": 0.001374046492855996, "learning_rate": 4.973512367388038e-06, "loss": 1.2746955690090545e-05, "num_tokens": 269087.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 56, "step_time": 7.920732168999962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 37.625, "completions/mean_terminated_length": 37.625, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.04022482968866825, "epoch": 0.4453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0068351044319570065, "kl": 0.0017894834163598716, "learning_rate": 4.970104766388833e-06, "loss": 1.8479673599358648e-05, "num_tokens": 273396.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 57, "step_time": 11.183577817000014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 23.25, "completions/mean_terminated_length": 23.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06758509948849678, "epoch": 0.453125, "frac_reward_zero_std": 0.5, "grad_norm": 1.572488784790039, "kl": 0.0030532198143191636, "learning_rate": 4.966492344651006e-06, "loss": -0.07791005074977875, "num_tokens": 278234.0, "reward": 0.8812500238418579, "reward_std": 0.3358757197856903, "rewards/reward_fn/mean": 0.8812500238418579, "rewards/reward_fn/std": 0.3358757197856903, "step": 58, "step_time": 7.65875285900006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.033227477222681046, "epoch": 0.4609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0016361080342903733, "kl": 0.0016750092036090791, "learning_rate": 4.962675401716056e-06, "loss": 1.6619302186882123e-05, "num_tokens": 282454.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 59, "step_time": 5.975613750999969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13986211270093918, "epoch": 0.46875, "frac_reward_zero_std": 1.0, "grad_norm": 0.04219949617981911, "kl": 0.004549495875835419, "learning_rate": 4.958654254084356e-06, "loss": 4.563133552437648e-05, "num_tokens": 287961.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 60, "step_time": 5.713278319999972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.25, "completions/mean_terminated_length": 21.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0659055095165968, "epoch": 0.4765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.010506045073270798, "kl": 0.0030614889692515135, "learning_rate": 4.954429235188897e-06, "loss": 2.8885815481771715e-05, "num_tokens": 292807.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 61, "step_time": 6.593735059999972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.040277909487485886, "epoch": 0.484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.006953485310077667, "kl": 0.0017460978706367314, "learning_rate": 4.95000069536765e-06, "loss": 1.6949392374954186e-05, "num_tokens": 296982.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 62, "step_time": 5.818303646000004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.14363406598567963, "epoch": 0.4921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0677562728524208, "kl": 0.00942042376846075, "learning_rate": 4.9453690018345144e-06, "loss": 9.429677447769791e-05, "num_tokens": 302489.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 63, "step_time": 5.593983074999983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12100231274962425, "epoch": 0.5, "frac_reward_zero_std": 1.0, "grad_norm": 0.01809072121977806, "kl": 0.003618443734012544, "learning_rate": 4.940534538648862e-06, "loss": 3.6511512007564306e-05, "num_tokens": 308019.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 64, "step_time": 5.77941624999994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03561065532267094, "epoch": 0.5078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0022499635815620422, "kl": 0.0012880651047453284, "learning_rate": 4.935497706683698e-06, "loss": 1.28087995108217e-05, "num_tokens": 312083.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 65, "step_time": 5.752014847000055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 15.25, "completions/mean_terminated_length": 15.25, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "entropy": 0.14075982943177223, "epoch": 0.515625, "frac_reward_zero_std": 0.5, "grad_norm": 2.6456313133239746, "kl": 0.09695499250665307, "learning_rate": 4.9302589235924185e-06, "loss": -0.08113337308168411, "num_tokens": 316841.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 66, "step_time": 6.563686877999999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07007651403546333, "epoch": 0.5234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.017309362068772316, "kl": 0.004948096117004752, "learning_rate": 4.924818623774178e-06, "loss": 4.838653694605455e-05, "num_tokens": 321630.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 67, "step_time": 6.570541892999927 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 23.375, "completions/mean_terminated_length": 23.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0408208854496479, "epoch": 0.53125, "frac_reward_zero_std": 1.0, "grad_norm": 0.012536097317934036, "kl": 0.004252667655237019, "learning_rate": 4.91917725833787e-06, "loss": 3.6156445275992155e-05, "num_tokens": 325569.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 68, "step_time": 5.749754669000026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 13.0, "completions/max_terminated_length": 13.0, "completions/mean_length": 13.0, "completions/mean_terminated_length": 13.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1359182596206665, "epoch": 0.5390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.03752928227186203, "kl": 0.009188917931169271, "learning_rate": 4.913335295064721e-06, "loss": 9.188917465507984e-05, "num_tokens": 331045.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 69, "step_time": 5.468208492999906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07358374819159508, "epoch": 0.546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.020383596420288086, "kl": 0.005584244150668383, "learning_rate": 4.907293218369499e-06, "loss": 5.529334521270357e-05, "num_tokens": 335761.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 70, "step_time": 6.64621212499992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.0, "completions/mean_terminated_length": 17.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05005803145468235, "epoch": 0.5546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.03125562518835068, "kl": 0.008841587696224451, "learning_rate": 4.901051529260352e-06, "loss": 8.841587987262756e-05, "num_tokens": 339757.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 71, "step_time": 5.899700159999952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 17.25, "completions/mean_terminated_length": 17.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07237722724676132, "epoch": 0.5625, "frac_reward_zero_std": 1.0, "grad_norm": 0.018481554463505745, "kl": 0.006117145298048854, "learning_rate": 4.89461074529726e-06, "loss": 6.117144948802888e-05, "num_tokens": 344555.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 72, "step_time": 6.860756025000001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.045137686654925346, "epoch": 0.5703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.020073302090168, "kl": 0.005907518556341529, "learning_rate": 4.8879714005491205e-06, "loss": 5.907518061576411e-05, "num_tokens": 348591.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 73, "step_time": 5.62259969999991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0720517449080944, "epoch": 0.578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.027030251920223236, "kl": 0.01036432757973671, "learning_rate": 4.881134045549463e-06, "loss": 8.4122279076837e-05, "num_tokens": 353312.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 74, "step_time": 6.612577034999958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 13.0, "completions/mean_length": 27.375, "completions/mean_terminated_length": 13.000000953674316, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1056247390806675, "epoch": 0.5859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.019467800855636597, "kl": 0.006749335676431656, "learning_rate": 4.874099247250799e-06, "loss": 5.878092997591011e-05, "num_tokens": 358931.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 75, "step_time": 14.855594667999867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.048843516036868095, "epoch": 0.59375, "frac_reward_zero_std": 1.0, "grad_norm": 0.033446282148361206, "kl": 0.012164841871708632, "learning_rate": 4.8668675889776095e-06, "loss": 0.00012165858061052859, "num_tokens": 362844.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 76, "step_time": 5.699464158999945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.0, "completions/mean_terminated_length": 17.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07759269699454308, "epoch": 0.6015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.017857957631349564, "kl": 0.0074398701544851065, "learning_rate": 4.85943967037798e-06, "loss": 6.409453635569662e-05, "num_tokens": 367560.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 77, "step_time": 6.886273725999672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 15.875, "completions/mean_terminated_length": 15.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12127615511417389, "epoch": 0.609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.024566544219851494, "kl": 0.007913533365353942, "learning_rate": 4.851816107373871e-06, "loss": 7.854383147787303e-05, "num_tokens": 373083.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 78, "step_time": 7.353863909999973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13099069893360138, "epoch": 0.6171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0477265790104866, "kl": 0.011370216961950064, "learning_rate": 4.843997532110051e-06, "loss": 0.00011370217544026673, "num_tokens": 378589.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 79, "step_time": 5.622442682000155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07205060124397278, "epoch": 0.625, "frac_reward_zero_std": 1.0, "grad_norm": 0.012449050322175026, "kl": 0.00409984530415386, "learning_rate": 4.835984592901678e-06, "loss": 3.9597347495146096e-05, "num_tokens": 383357.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 80, "step_time": 6.542991357999881 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 21.875, "completions/mean_terminated_length": 21.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07283683493733406, "epoch": 0.6328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.019634408876299858, "kl": 0.007241100538522005, "learning_rate": 4.82777795418054e-06, "loss": 7.439294131472707e-05, "num_tokens": 388152.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 81, "step_time": 7.132434741999987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 15.25, "completions/mean_terminated_length": 15.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08575013279914856, "epoch": 0.640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.03768326714634895, "kl": 0.010142359882593155, "learning_rate": 4.819378296439962e-06, "loss": 0.00010288630437571555, "num_tokens": 392866.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 82, "step_time": 6.954328571000133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 13.0, "completions/max_terminated_length": 13.0, "completions/mean_length": 13.0, "completions/mean_terminated_length": 13.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12997420877218246, "epoch": 0.6484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.01888282783329487, "kl": 0.005712541053071618, "learning_rate": 4.810786316178377e-06, "loss": 5.712541315006092e-05, "num_tokens": 398394.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 83, "step_time": 6.931600935000006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 22.125, "completions/mean_terminated_length": 22.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06609366461634636, "epoch": 0.65625, "frac_reward_zero_std": 1.0, "grad_norm": 0.017994992434978485, "kl": 0.00510314735583961, "learning_rate": 4.802002725841577e-06, "loss": 4.8029709432739764e-05, "num_tokens": 403295.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 84, "step_time": 7.442147588000125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 18.625, "completions/mean_terminated_length": 18.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11900537833571434, "epoch": 0.6640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.024866754189133644, "kl": 0.006785314995795488, "learning_rate": 4.793028253763633e-06, "loss": 7.137990178307518e-05, "num_tokens": 408052.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 85, "step_time": 7.643961140000329 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10917261242866516, "epoch": 0.671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.022167593240737915, "kl": 0.0044796590227633715, "learning_rate": 4.783863644106502e-06, "loss": 4.586868089972995e-05, "num_tokens": 413581.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 86, "step_time": 7.9875519340000665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06199129670858383, "epoch": 0.6796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.03173369914293289, "kl": 0.008890938945114613, "learning_rate": 4.774509656798326e-06, "loss": 8.7326108769048e-05, "num_tokens": 418435.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 87, "step_time": 6.503118833999906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0360421109944582, "epoch": 0.6875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01023577805608511, "kl": 0.003528562723658979, "learning_rate": 4.764967067470409e-06, "loss": 3.279188240412623e-05, "num_tokens": 422510.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 88, "step_time": 5.7346701770002255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13792025297880173, "epoch": 0.6953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.036804474890232086, "kl": 0.007783198030665517, "learning_rate": 4.755236667392914e-06, "loss": 7.783197361277416e-05, "num_tokens": 428016.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 89, "step_time": 5.881843299000138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.033070383593440056, "epoch": 0.703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.002256783191114664, "kl": 0.001209279813338071, "learning_rate": 4.745319263409241e-06, "loss": 1.2092797987861559e-05, "num_tokens": 432116.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 90, "step_time": 5.656249395999794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 36.25, "completions/mean_terminated_length": 23.142858505249023, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.2254822440445423, "epoch": 0.7109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.014093738049268723, "kl": 0.0030488879419863224, "learning_rate": 4.735215677869129e-06, "loss": 3.228533023502678e-05, "num_tokens": 437806.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 91, "step_time": 14.648245948000067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 26.75, "completions/mean_terminated_length": 26.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05311024375259876, "epoch": 0.71875, "frac_reward_zero_std": 1.0, "grad_norm": 0.00612290482968092, "kl": 0.0022808221401646733, "learning_rate": 4.724926748560464e-06, "loss": 2.2808220819570124e-05, "num_tokens": 442648.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 92, "step_time": 7.179930319999812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11795984953641891, "epoch": 0.7265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.014634083956480026, "kl": 0.0032170950435101986, "learning_rate": 4.714453328639814e-06, "loss": 3.587050741771236e-05, "num_tokens": 448179.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 93, "step_time": 7.483695861999877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 19.875, "completions/mean_terminated_length": 19.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07148051261901855, "epoch": 0.734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.010934879072010517, "kl": 0.0029092643526382744, "learning_rate": 4.7037962865616795e-06, "loss": 2.7483838493935764e-05, "num_tokens": 453030.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 94, "step_time": 7.076518674999988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 22.125, "completions/mean_terminated_length": 22.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06901054829359055, "epoch": 0.7421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.006595511920750141, "kl": 0.0014058846863918006, "learning_rate": 4.692956506006486e-06, "loss": 1.403903206664836e-05, "num_tokens": 457855.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 95, "step_time": 7.476475853000011 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08090204373002052, "epoch": 0.75, "frac_reward_zero_std": 1.0, "grad_norm": 0.020713720470666885, "kl": 0.005518489051610231, "learning_rate": 4.681934885807307e-06, "loss": 5.53158279217314e-05, "num_tokens": 462588.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 96, "step_time": 6.740178639000078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.037110449746251106, "epoch": 0.7578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.006115948315709829, "kl": 0.0022233356721699238, "learning_rate": 4.6707323398753346e-06, "loss": 2.206304998253472e-05, "num_tokens": 466604.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 97, "step_time": 5.532072194999955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06933313608169556, "epoch": 0.765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007001116406172514, "kl": 0.0019759326823987067, "learning_rate": 4.659349797124096e-06, "loss": 1.9955001334892586e-05, "num_tokens": 471382.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 98, "step_time": 6.4825494700000945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.5, "completions/mean_terminated_length": 25.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03348589688539505, "epoch": 0.7734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.002566123381257057, "kl": 0.0018573449924588203, "learning_rate": 4.647788201392429e-06, "loss": 1.857344977906905e-05, "num_tokens": 475554.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 99, "step_time": 5.890160061000188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 128.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 42.0, "completions/mean_terminated_length": 13.333333969116211, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.21250910684466362, "epoch": 0.78125, "frac_reward_zero_std": 1.0, "grad_norm": 0.07006637006998062, "kl": 0.017085027415305376, "learning_rate": 4.636048511366222e-06, "loss": 0.00017085025319829583, "num_tokens": 481290.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 100, "step_time": 14.571886820999907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.25, "completions/mean_terminated_length": 16.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11931024491786957, "epoch": 0.7890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0073622651398181915, "kl": 0.0012832598586101085, "learning_rate": 4.624131700498913e-06, "loss": 1.3542096894525457e-05, "num_tokens": 486820.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 101, "step_time": 7.506882940999958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 33.75, "completions/mean_terminated_length": 20.285715103149414, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05769490636885166, "epoch": 0.796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.002226189011707902, "kl": 0.001102528884075582, "learning_rate": 4.612038756930778e-06, "loss": 8.816463378025219e-06, "num_tokens": 491814.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 102, "step_time": 14.50194374800003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.375, "completions/mean_terminated_length": 27.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.032190028578042984, "epoch": 0.8046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0019165624398738146, "kl": 0.0013525813119485974, "learning_rate": 4.599770683406992e-06, "loss": 1.3399311683315318e-05, "num_tokens": 495785.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 103, "step_time": 5.6954472240001905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 22.125, "completions/mean_terminated_length": 22.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09382087737321854, "epoch": 0.8125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0029365946538746357, "kl": 0.002741978387348354, "learning_rate": 4.587328497194478e-06, "loss": 2.726353341131471e-05, "num_tokens": 501338.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 104, "step_time": 7.473332564999964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 26.75, "completions/mean_terminated_length": 26.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05792428180575371, "epoch": 0.8203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0022111833095550537, "kl": 0.0035839949268847704, "learning_rate": 4.5747132299975634e-06, "loss": 4.1183855501003563e-05, "num_tokens": 506124.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 105, "step_time": 7.173797810999986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.625, "completions/mean_terminated_length": 21.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06740124337375164, "epoch": 0.828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0050131166353821754, "kl": 0.0017226869240403175, "learning_rate": 4.561925927872421e-06, "loss": 1.599166716914624e-05, "num_tokens": 510953.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 106, "step_time": 6.916509818999884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07052117213606834, "epoch": 0.8359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.007340835873037577, "kl": 0.0017220252193510532, "learning_rate": 4.548967651140341e-06, "loss": 1.7247453797608614e-05, "num_tokens": 515806.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 107, "step_time": 6.615679707000027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 24.125, "completions/mean_terminated_length": 24.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06415174063295126, "epoch": 0.84375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0005465157446451485, "kl": 0.001015377274597995, "learning_rate": 4.5358394742998e-06, "loss": 1.1552911928447429e-05, "num_tokens": 520699.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 108, "step_time": 7.493365885000003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.25, "completions/mean_terminated_length": 21.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0641067698597908, "epoch": 0.8515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.010530305095016956, "kl": 0.0026437516789883375, "learning_rate": 4.522542485937369e-06, "loss": 2.6550886104814708e-05, "num_tokens": 525437.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 109, "step_time": 6.527635427999712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 25.75, "completions/mean_terminated_length": 25.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0577393751591444, "epoch": 0.859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.008794880472123623, "kl": 0.0028055842267349362, "learning_rate": 4.509077788637446e-06, "loss": 2.7364983907318674e-05, "num_tokens": 530267.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 110, "step_time": 7.3437313919998815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.0, "completions/mean_terminated_length": 24.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0586474034935236, "epoch": 0.8671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.005623109173029661, "kl": 0.0036008708993904293, "learning_rate": 4.4954464988908306e-06, "loss": 3.5230626963311806e-05, "num_tokens": 535131.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 111, "step_time": 7.285259039999801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.375, "completions/mean_terminated_length": 16.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12351097911596298, "epoch": 0.875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01802302524447441, "kl": 0.0029585817828774452, "learning_rate": 4.481649747002146e-06, "loss": 3.03143824567087e-05, "num_tokens": 540638.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 112, "step_time": 7.425657803999911 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06710582785308361, "epoch": 0.8828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0028362085577100515, "kl": 0.0015989196253940463, "learning_rate": 4.467688676996111e-06, "loss": 1.6505855455761775e-05, "num_tokens": 545519.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 113, "step_time": 6.964208683000152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07923074252903461, "epoch": 0.890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.008551168255507946, "kl": 0.002287313051056117, "learning_rate": 4.4535644465226795e-06, "loss": 1.9634611817309633e-05, "num_tokens": 550319.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 114, "step_time": 6.80012675099988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 20.0, "completions/mean_terminated_length": 20.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07435985282063484, "epoch": 0.8984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.014601176604628563, "kl": 0.005440297303721309, "learning_rate": 4.43927822676105e-06, "loss": 4.615134821506217e-05, "num_tokens": 555091.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 115, "step_time": 7.212101119999943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08148583956062794, "epoch": 0.90625, "frac_reward_zero_std": 1.0, "grad_norm": 0.004445843398571014, "kl": 0.001686684088781476, "learning_rate": 4.424831202322548e-06, "loss": 1.5020732462289743e-05, "num_tokens": 559873.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 116, "step_time": 6.573965233000081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 30.875, "completions/mean_terminated_length": 30.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05395838990807533, "epoch": 0.9140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.002361060120165348, "kl": 0.0016205881838686764, "learning_rate": 4.410224571152402e-06, "loss": 1.6156718629645184e-05, "num_tokens": 564700.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 117, "step_time": 7.449374204000151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.5, "completions/mean_terminated_length": 25.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03191390447318554, "epoch": 0.921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.04630071669816971, "kl": 0.015495602798182517, "learning_rate": 4.395459544430407e-06, "loss": 0.00015495602565351874, "num_tokens": 568832.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 118, "step_time": 5.798940031999791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08851146325469017, "epoch": 0.9296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.03349859267473221, "kl": 0.0069463985855691135, "learning_rate": 4.380537346470495e-06, "loss": 7.56498338887468e-05, "num_tokens": 573559.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 119, "step_time": 6.745357660999844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.041911693289875984, "epoch": 0.9375, "frac_reward_zero_std": 1.0, "grad_norm": 0.003103650640696287, "kl": 0.0021437766263261437, "learning_rate": 4.3654592146192146e-06, "loss": 2.1203726646490395e-05, "num_tokens": 577735.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 120, "step_time": 5.838892162999855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.5, "completions/mean_terminated_length": 24.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0581835713237524, "epoch": 0.9453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0017389818094670773, "kl": 0.00121387725812383, "learning_rate": 4.35022639915313e-06, "loss": 1.2284486729186028e-05, "num_tokens": 582627.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 121, "step_time": 7.239668368000139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.027326339855790138, "epoch": 0.953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.00015551889373455197, "kl": 0.0014788485132157803, "learning_rate": 4.334840163175152e-06, "loss": 1.4788485714234412e-05, "num_tokens": 586899.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 122, "step_time": 5.908869453000079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08600620739161968, "epoch": 0.9609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.004444346763193607, "kl": 0.0013841381878592074, "learning_rate": 4.319301782509794e-06, "loss": 1.4433127944357693e-05, "num_tokens": 591680.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 123, "step_time": 6.542298487999915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 29.75, "completions/mean_terminated_length": 29.75, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.05058911070227623, "epoch": 0.96875, "frac_reward_zero_std": 1.0, "grad_norm": 0.002049674978479743, "kl": 0.0011390995350666344, "learning_rate": 4.30361254559739e-06, "loss": 1.1433874533395283e-05, "num_tokens": 596474.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 124, "step_time": 7.415240068999992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 20.25, "completions/mean_terminated_length": 20.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.14710266143083572, "epoch": 0.9765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.02711130492389202, "kl": 0.0030616128933615983, "learning_rate": 4.287773753387249e-06, "loss": 3.7990492273820564e-05, "num_tokens": 602012.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 125, "step_time": 7.445253116999993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 25.625, "completions/mean_terminated_length": 25.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.060312340036034584, "epoch": 0.984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.004181354772299528, "kl": 0.0010917744366452098, "learning_rate": 4.271786719229787e-06, "loss": 1.0562279385339934e-05, "num_tokens": 606797.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 126, "step_time": 7.632959243999949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 27.75, "completions/mean_terminated_length": 13.428571701049805, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10784962400794029, "epoch": 0.9921875, "frac_reward_zero_std": 1.0, "grad_norm": 1.5359536409378052, "kl": 0.055240524452528916, "learning_rate": 4.255652768767619e-06, "loss": 0.0008335658931173384, "num_tokens": 612419.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 127, "step_time": 14.956502908999937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 32.125, "completions/mean_terminated_length": 32.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09433136507868767, "epoch": 1.0, "frac_reward_zero_std": 1.0, "grad_norm": 0.026125041767954826, "kl": 0.005883891310077161, "learning_rate": 4.23937323982564e-06, "loss": 6.240967923076823e-05, "num_tokens": 617328.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 128, "step_time": 8.913310376000027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03300720080733299, "epoch": 1.0078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0028305482119321823, "kl": 0.001693042810074985, "learning_rate": 4.222949482300094e-06, "loss": 1.6930427591432817e-05, "num_tokens": 621356.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 129, "step_time": 5.800293608999937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07437415793538094, "epoch": 1.015625, "frac_reward_zero_std": 0.5, "grad_norm": 3.94826078414917, "kl": 2.5161777958273888, "learning_rate": 4.206382858046636e-06, "loss": -0.11797107756137848, "num_tokens": 626170.0, "reward": 0.887499988079071, "reward_std": 0.3181980550289154, "rewards/reward_fn/mean": 0.887499988079071, "rewards/reward_fn/std": 0.3181980550289154, "step": 130, "step_time": 6.945177734999788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 17.375, "completions/mean_terminated_length": 17.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07938191294670105, "epoch": 1.0234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.00952488835901022, "kl": 0.001886414596810937, "learning_rate": 4.189674740767411e-06, "loss": 1.6817120922496542e-05, "num_tokens": 630937.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 131, "step_time": 6.6918233229998805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.25, "completions/mean_terminated_length": 17.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07850120961666107, "epoch": 1.03125, "frac_reward_zero_std": 1.0, "grad_norm": 0.013484103605151176, "kl": 0.0045427161967381835, "learning_rate": 4.172826515897146e-06, "loss": 4.5427161239786074e-05, "num_tokens": 635691.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 132, "step_time": 6.658873882999842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08399690128862858, "epoch": 1.0390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.025045400485396385, "kl": 0.029486034996807575, "learning_rate": 4.15583958048827e-06, "loss": 0.00029239041032269597, "num_tokens": 640561.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 133, "step_time": 6.894192197999928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.125, "completions/mean_terminated_length": 13.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13748392462730408, "epoch": 1.046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01748264580965042, "kl": 0.001855719368904829, "learning_rate": 4.138715343095069e-06, "loss": 1.858307223301381e-05, "num_tokens": 646038.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 134, "step_time": 5.438569751999921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.125, "completions/mean_terminated_length": 23.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.04659535735845566, "epoch": 1.0546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.051196254789829254, "kl": 0.01883502001874149, "learning_rate": 4.12145522365689e-06, "loss": 0.0001754522672854364, "num_tokens": 650127.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 135, "step_time": 5.736987513000258 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07493950799107552, "epoch": 1.0625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007082380820065737, "kl": 0.0019909495022147894, "learning_rate": 4.104060653380403e-06, "loss": 2.0618410417228006e-05, "num_tokens": 654983.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 136, "step_time": 6.561122189999878 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07711902260780334, "epoch": 1.0703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.061376169323921204, "kl": 0.021036528050899506, "learning_rate": 4.086533074620919e-06, "loss": 0.0002116656833095476, "num_tokens": 659836.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 137, "step_time": 6.81616649800003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.038892749696969986, "epoch": 1.078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.005187615752220154, "kl": 0.0026534507051110268, "learning_rate": 4.068873940762796e-06, "loss": 2.6154470106121153e-05, "num_tokens": 664044.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 138, "step_time": 5.881332467999982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 15.625, "completions/mean_terminated_length": 15.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1011301763355732, "epoch": 1.0859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.2945476174354553, "kl": 0.08709336444735527, "learning_rate": 4.051084716098921e-06, "loss": 0.0008627058705314994, "num_tokens": 668817.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 139, "step_time": 6.639636123999935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.625, "completions/mean_terminated_length": 13.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12177715450525284, "epoch": 1.09375, "frac_reward_zero_std": 1.0, "grad_norm": 0.033345598727464676, "kl": 0.005186434369534254, "learning_rate": 4.033166875709291e-06, "loss": 5.185226473258808e-05, "num_tokens": 674298.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 140, "step_time": 5.517028535000009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.5, "completions/mean_terminated_length": 17.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07133128866553307, "epoch": 1.1015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.12929074466228485, "kl": 0.09029890596866608, "learning_rate": 4.015121905338704e-06, "loss": 0.0009029890061356127, "num_tokens": 678246.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 141, "step_time": 5.957433755000011 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.057012878358364105, "epoch": 1.109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0789097398519516, "kl": 0.03356556943617761, "learning_rate": 3.996951301273556e-06, "loss": 0.0003683689865283668, "num_tokens": 682300.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 142, "step_time": 5.954758885000047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07221270352602005, "epoch": 1.1171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.08531598001718521, "kl": 0.04459764843340963, "learning_rate": 3.9786565702177725e-06, "loss": 0.00040393066592514515, "num_tokens": 687145.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 143, "step_time": 6.8996876880000855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.14312979578971863, "epoch": 1.125, "frac_reward_zero_std": 1.0, "grad_norm": 0.020980026572942734, "kl": 0.03795890283072367, "learning_rate": 3.960239229167869e-06, "loss": 0.0003864379250444472, "num_tokens": 692651.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 144, "step_time": 5.689572013000088 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.125, "completions/mean_terminated_length": 13.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13573984801769257, "epoch": 1.1328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.014293034560978413, "kl": 0.003559689619578421, "learning_rate": 3.941700805287169e-06, "loss": 3.578899850253947e-05, "num_tokens": 698156.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 145, "step_time": 5.635921549999921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.75, "completions/mean_terminated_length": 19.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07221874222159386, "epoch": 1.140625, "frac_reward_zero_std": 0.5, "grad_norm": 2.1729769706726074, "kl": 0.006373312848154455, "learning_rate": 3.92304283577916e-06, "loss": -0.15180744230747223, "num_tokens": 703038.0, "reward": 0.8812500238418579, "reward_std": 0.3358757197856903, "rewards/reward_fn/mean": 0.8812500238418579, "rewards/reward_fn/std": 0.3358757197856903, "step": 146, "step_time": 6.94864533499981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06784151494503021, "epoch": 1.1484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.008188405074179173, "kl": 0.005511927942279726, "learning_rate": 3.904266867760044e-06, "loss": 5.116416286909953e-05, "num_tokens": 707899.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 147, "step_time": 6.646591747999992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07835190743207932, "epoch": 1.15625, "frac_reward_zero_std": 1.0, "grad_norm": 0.004332108423113823, "kl": 0.0011774248559959233, "learning_rate": 3.8853744581304376e-06, "loss": 1.2425527529558167e-05, "num_tokens": 712639.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 148, "step_time": 6.73792784200009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07140998169779778, "epoch": 1.1640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.006420095916837454, "kl": 0.008202763739973307, "learning_rate": 3.866367173446281e-06, "loss": 7.896333409007639e-05, "num_tokens": 717406.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 149, "step_time": 6.581852653999931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07744136825203896, "epoch": 1.171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.006486960221081972, "kl": 0.0021612545242533088, "learning_rate": 3.84724658978894e-06, "loss": 2.1575528080575168e-05, "num_tokens": 722111.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 150, "step_time": 6.601302765000128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.5, "completions/mean_terminated_length": 13.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12781532481312752, "epoch": 1.1796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.011086604557931423, "kl": 0.0017027303110808134, "learning_rate": 3.828014292634508e-06, "loss": 1.7027303329086863e-05, "num_tokens": 727595.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 151, "step_time": 5.656125526000324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.75, "completions/mean_terminated_length": 19.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07872535288333893, "epoch": 1.1875, "frac_reward_zero_std": 1.0, "grad_norm": 0.030335335060954094, "kl": 0.017175353597849607, "learning_rate": 3.808671876722357e-06, "loss": 0.00016482984938193113, "num_tokens": 732313.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 152, "step_time": 6.728792943000144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 23.25, "completions/mean_terminated_length": 23.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06778555735945702, "epoch": 1.1953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.005629899445921183, "kl": 0.001822992053348571, "learning_rate": 3.7892209459228802e-06, "loss": 1.8229919078294188e-05, "num_tokens": 737211.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 153, "step_time": 7.577010963999783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03097234107553959, "epoch": 1.203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.00039684277726337314, "kl": 0.0014657879946753383, "learning_rate": 3.769663113104516e-06, "loss": 1.4650184311904013e-05, "num_tokens": 741306.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 154, "step_time": 5.836096522999696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.75, "completions/mean_terminated_length": 13.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1351180225610733, "epoch": 1.2109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.04005980119109154, "kl": 0.012403905857354403, "learning_rate": 3.7500000000000005e-06, "loss": 0.00012463353050407022, "num_tokens": 746788.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 155, "step_time": 5.403199547000213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 25.25, "completions/mean_terminated_length": 25.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05965032987296581, "epoch": 1.21875, "frac_reward_zero_std": 1.0, "grad_norm": 0.007664468139410019, "kl": 0.0025948547408916056, "learning_rate": 3.7302332370718988e-06, "loss": 2.4826764274621382e-05, "num_tokens": 751706.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 156, "step_time": 7.5867298230000415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 23.375, "completions/mean_terminated_length": 23.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03286047466099262, "epoch": 1.2265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0014668001094833016, "kl": 0.0018113566911779344, "learning_rate": 3.7103644633774015e-06, "loss": 1.7852235032478347e-05, "num_tokens": 755809.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 157, "step_time": 5.844333027999937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.0, "completions/mean_terminated_length": 17.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07944399863481522, "epoch": 1.234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.00917236041277647, "kl": 0.0026575025403872132, "learning_rate": 3.690395326432421e-06, "loss": 2.6575022275210358e-05, "num_tokens": 760533.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 158, "step_time": 6.835314456999868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07627736404538155, "epoch": 1.2421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.016258908435702324, "kl": 0.003968668053857982, "learning_rate": 3.6703274820749736e-06, "loss": 3.887472485075705e-05, "num_tokens": 765320.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 159, "step_time": 6.874671181000167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06996462866663933, "epoch": 1.25, "frac_reward_zero_std": 1.0, "grad_norm": 0.011779413558542728, "kl": 0.002461543306708336, "learning_rate": 3.650162594327881e-06, "loss": 2.4944250981207006e-05, "num_tokens": 770199.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 160, "step_time": 6.942916448000005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 29.5, "completions/mean_terminated_length": 29.5, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.02681659162044525, "epoch": 1.2578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.00040976243326440454, "kl": 0.0014975924859754741, "learning_rate": 3.6299023352607894e-06, "loss": 1.4975924386817496e-05, "num_tokens": 774395.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 161, "step_time": 6.015989045999959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 26.875, "completions/mean_terminated_length": 26.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.061081623658537865, "epoch": 1.265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.014576679095625877, "kl": 0.0038794887950643897, "learning_rate": 3.6095483848515223e-06, "loss": 3.555541479727253e-05, "num_tokens": 779170.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 162, "step_time": 7.245200251999904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 22.125, "completions/mean_terminated_length": 22.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06859908811748028, "epoch": 1.2734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.020452730357646942, "kl": 0.005623190430924296, "learning_rate": 3.589102430846773e-06, "loss": 5.230373062659055e-05, "num_tokens": 784051.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 163, "step_time": 7.287538059000099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.028452389873564243, "epoch": 1.28125, "frac_reward_zero_std": 1.0, "grad_norm": 0.004462169948965311, "kl": 0.001680202956777066, "learning_rate": 3.5685661686221644e-06, "loss": 1.6027215679059736e-05, "num_tokens": 788131.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 164, "step_time": 5.703693580999925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05600108206272125, "epoch": 1.2890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.021526971831917763, "kl": 0.006003132904879749, "learning_rate": 3.5479413010416606e-06, "loss": 5.7515826483722776e-05, "num_tokens": 792967.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 165, "step_time": 6.575796513000114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.02504791971296072, "epoch": 1.296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0030265813693404198, "kl": 0.0021526971249841154, "learning_rate": 3.527229538316371e-06, "loss": 2.115210190822836e-05, "num_tokens": 797183.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 166, "step_time": 5.972162173999777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13807643204927444, "epoch": 1.3046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.03702490031719208, "kl": 0.006319678155705333, "learning_rate": 3.5064325978627365e-06, "loss": 6.32827723165974e-05, "num_tokens": 802710.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 167, "step_time": 5.624996752000243 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.875, "completions/mean_terminated_length": 27.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.02927943505346775, "epoch": 1.3125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0027886745519936085, "kl": 0.001402048219460994, "learning_rate": 3.4855522041601265e-06, "loss": 1.3811075405101292e-05, "num_tokens": 806769.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 168, "step_time": 5.729365316999974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 22.875, "completions/mean_terminated_length": 22.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09153914824128151, "epoch": 1.3203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.031587954610586166, "kl": 0.006058453815057874, "learning_rate": 3.4645900886078388e-06, "loss": 6.320517422864214e-05, "num_tokens": 812348.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 169, "step_time": 7.63299943700008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.5, "completions/mean_terminated_length": 21.5, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.05722755752503872, "epoch": 1.328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.014601275324821472, "kl": 0.003756172431167215, "learning_rate": 3.443547989381536e-06, "loss": 3.436290717218071e-05, "num_tokens": 817204.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 170, "step_time": 6.65200465199996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09964188188314438, "epoch": 1.3359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.06053476780653, "kl": 0.006807157536968589, "learning_rate": 3.422427651289118e-06, "loss": 6.788011523894966e-05, "num_tokens": 822734.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 171, "step_time": 7.496788298999945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 29.625, "completions/mean_terminated_length": 29.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05451471544802189, "epoch": 1.34375, "frac_reward_zero_std": 1.0, "grad_norm": 0.008337209932506084, "kl": 0.0015613330178894103, "learning_rate": 3.4012308256260366e-06, "loss": 1.5506457202718593e-05, "num_tokens": 827671.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 172, "step_time": 7.244489809999777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 38.875, "completions/mean_terminated_length": 26.142858505249023, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.6550341844558716, "epoch": 1.3515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.008499568328261375, "kl": 0.0018006233731284738, "learning_rate": 3.3799592700300867e-06, "loss": 1.8337512301513925e-05, "num_tokens": 833354.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 173, "step_time": 14.681682713999862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 23.875, "completions/mean_terminated_length": 23.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.060140494257211685, "epoch": 1.359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.00928256381303072, "kl": 0.0029123042477294803, "learning_rate": 3.3586147483356534e-06, "loss": 2.732695429585874e-05, "num_tokens": 838109.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 174, "step_time": 7.2412520599998516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 24.375, "completions/mean_terminated_length": 24.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06309142522513866, "epoch": 1.3671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.00517475139349699, "kl": 0.0014793115551583469, "learning_rate": 3.3371990304274654e-06, "loss": 1.494552179792663e-05, "num_tokens": 843028.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 175, "step_time": 7.81271701799983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03291504085063934, "epoch": 1.375, "frac_reward_zero_std": 1.0, "grad_norm": 0.004775932524353266, "kl": 0.002072959439828992, "learning_rate": 3.315713892093829e-06, "loss": 2.0729594325530343e-05, "num_tokens": 847064.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 176, "step_time": 5.776862065000159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 16.25, "completions/mean_terminated_length": 16.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13138685747981071, "epoch": 1.3828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.023814553394913673, "kl": 0.0031949164113029838, "learning_rate": 3.294161114879382e-06, "loss": 3.185410605510697e-05, "num_tokens": 852594.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 177, "step_time": 7.431254295000144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.25, "completions/mean_terminated_length": 16.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12037009745836258, "epoch": 1.390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.018406132236123085, "kl": 0.003100107191130519, "learning_rate": 3.272542485937369e-06, "loss": 3.313878914923407e-05, "num_tokens": 858124.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 178, "step_time": 7.8139032770000085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 28.625, "completions/mean_terminated_length": 28.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.16150045860558748, "epoch": 1.3984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.24875901639461517, "kl": 0.045317163807339966, "learning_rate": 3.2508597978814515e-06, "loss": 0.00043976842425763607, "num_tokens": 862249.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 179, "step_time": 6.564737340999727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03088196087628603, "epoch": 1.40625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0007541459635831416, "kl": 0.0011516560916788876, "learning_rate": 3.2291148486370626e-06, "loss": 1.1446018106653355e-05, "num_tokens": 866265.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 180, "step_time": 5.712320224999758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.75, "completions/mean_terminated_length": 21.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06882932037115097, "epoch": 1.4140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.00206294865347445, "kl": 0.0013715573586523533, "learning_rate": 3.207309441292325e-06, "loss": 1.3715573004446924e-05, "num_tokens": 871011.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 181, "step_time": 6.726993576000041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 29.125, "completions/mean_terminated_length": 29.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.5393692404031754, "epoch": 1.421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.02723606675863266, "kl": 0.0031313110375776887, "learning_rate": 3.185445383948539e-06, "loss": 3.407296753721312e-05, "num_tokens": 876640.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 182, "step_time": 11.819733140000153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 20.75, "completions/mean_terminated_length": 20.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12167311646044254, "epoch": 1.4296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.02585308626294136, "kl": 0.002457446011248976, "learning_rate": 3.1635244895702527e-06, "loss": 2.4284261598950252e-05, "num_tokens": 881406.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 183, "step_time": 6.903921601999855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06957883015275002, "epoch": 1.4375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0076453592628240585, "kl": 0.001370629877783358, "learning_rate": 3.1415485758349344e-06, "loss": 1.3765486073680222e-05, "num_tokens": 886226.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 184, "step_time": 7.334037624999837 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0735643245279789, "epoch": 1.4453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.005766255781054497, "kl": 0.0016721467836759984, "learning_rate": 3.11951946498225e-06, "loss": 1.5945741324685514e-05, "num_tokens": 890954.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 185, "step_time": 6.7520039959997575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 33.375, "completions/mean_terminated_length": 33.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.043096503242850304, "epoch": 1.453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.004805476404726505, "kl": 0.001356584718450904, "learning_rate": 3.0974389836629628e-06, "loss": 1.3770947589364368e-05, "num_tokens": 895109.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 186, "step_time": 9.572878103999756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.375, "completions/mean_terminated_length": 16.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1160760298371315, "epoch": 1.4609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.021316073834896088, "kl": 0.0027994479751214385, "learning_rate": 3.0753089627874668e-06, "loss": 2.722722274484113e-05, "num_tokens": 900640.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 187, "step_time": 7.767691414999717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.034027645364403725, "epoch": 1.46875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0012773608323186636, "kl": 0.0015459859278053045, "learning_rate": 3.0531312373739695e-06, "loss": 1.545986015116796e-05, "num_tokens": 904644.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 188, "step_time": 5.866847418999896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 31.75, "completions/mean_terminated_length": 31.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05926687642931938, "epoch": 1.4765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.02531527541577816, "kl": 0.0033996477723121643, "learning_rate": 3.030907646396333e-06, "loss": 3.2442520023323596e-05, "num_tokens": 909574.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 189, "step_time": 11.722558253999978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 21.875, "completions/mean_terminated_length": 21.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06604529172182083, "epoch": 1.484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.014295347966253757, "kl": 0.004741971264593303, "learning_rate": 3.0086400326315853e-06, "loss": 4.750135849462822e-05, "num_tokens": 914337.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 190, "step_time": 6.908794395000086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06935635954141617, "epoch": 1.4921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0014209807850420475, "kl": 0.0011486579023767263, "learning_rate": 2.9863302425071156e-06, "loss": 1.2237365808687173e-05, "num_tokens": 919185.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 191, "step_time": 6.616372841999919 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 18.75, "completions/mean_terminated_length": 18.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11195787042379379, "epoch": 1.5, "frac_reward_zero_std": 1.0, "grad_norm": 0.11325943470001221, "kl": 0.004804897769645322, "learning_rate": 2.963980125947573e-06, "loss": 6.231457518879324e-05, "num_tokens": 924735.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 192, "step_time": 7.530646898999976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 15.875, "completions/mean_terminated_length": 15.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1282949000597, "epoch": 1.5078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.02611485868692398, "kl": 0.001775389551767148, "learning_rate": 2.941591536221469e-06, "loss": 2.0562109057209454e-05, "num_tokens": 930262.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 193, "step_time": 7.432371278000119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.029940275475382805, "epoch": 1.515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0013159937225282192, "kl": 0.001477666082791984, "learning_rate": 2.9191663297875027e-06, "loss": 1.4661342902400065e-05, "num_tokens": 934362.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 194, "step_time": 5.929965283999991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.5, "completions/mean_terminated_length": 13.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.142668217420578, "epoch": 1.5234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.05234305188059807, "kl": 0.005082390503957868, "learning_rate": 2.896706366140629e-06, "loss": 5.082390271127224e-05, "num_tokens": 939870.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 195, "step_time": 5.839425092999818 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06505489535629749, "epoch": 1.53125, "frac_reward_zero_std": 1.0, "grad_norm": 0.002731472020968795, "kl": 0.001581202377565205, "learning_rate": 2.8742135076578608e-06, "loss": 1.5823465219000354e-05, "num_tokens": 944735.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 196, "step_time": 6.905690054999923 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11012165620923042, "epoch": 1.5390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0070010521449148655, "kl": 0.0010562781826592982, "learning_rate": 2.8516896194438515e-06, "loss": 1.0670170013327152e-05, "num_tokens": 950290.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 197, "step_time": 7.652169488000027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 21.875, "completions/mean_terminated_length": 21.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06521525606513023, "epoch": 1.546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.017679838463664055, "kl": 0.0035706963972188532, "learning_rate": 2.8291365691762313e-06, "loss": 3.664140967885032e-05, "num_tokens": 955021.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 198, "step_time": 7.155417588999853 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.25, "completions/mean_terminated_length": 25.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.049559252336621284, "epoch": 1.5546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.05125805735588074, "kl": 0.01641816832125187, "learning_rate": 2.8065562269507464e-06, "loss": 0.00016517679614480585, "num_tokens": 959027.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 199, "step_time": 5.624229268999898 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 26.875, "completions/mean_terminated_length": 26.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.04596647992730141, "epoch": 1.5625, "frac_reward_zero_std": 1.0, "grad_norm": 0.009446537122130394, "kl": 0.0029358353931456804, "learning_rate": 2.7839504651261873e-06, "loss": 3.024380566785112e-05, "num_tokens": 963106.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 200, "step_time": 5.7302313129998765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 24.125, "completions/mean_terminated_length": 24.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06624070927500725, "epoch": 1.5703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0009546372457407415, "kl": 0.0009917069983202964, "learning_rate": 2.761321158169134e-06, "loss": 1.0749818102340214e-05, "num_tokens": 968011.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 201, "step_time": 7.5783325940001305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.125, "completions/mean_terminated_length": 23.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03553210385143757, "epoch": 1.578125, "frac_reward_zero_std": 0.5, "grad_norm": 1.7552725076675415, "kl": 0.06742474652128294, "learning_rate": 2.7386701824985257e-06, "loss": -0.07762715220451355, "num_tokens": 972088.0, "reward": 0.8812500238418579, "reward_std": 0.3358757197856903, "rewards/reward_fn/mean": 0.8812500238418579, "rewards/reward_fn/std": 0.3358757197856903, "step": 202, "step_time": 5.655839926999988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06611593812704086, "epoch": 1.5859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.002574133686721325, "kl": 0.0014704846544191241, "learning_rate": 2.715999416330068e-06, "loss": 1.4717452359036542e-05, "num_tokens": 976941.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 203, "step_time": 6.626762887999803 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.375, "completions/mean_terminated_length": 16.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12001239135861397, "epoch": 1.59375, "frac_reward_zero_std": 1.0, "grad_norm": 0.022112533450126648, "kl": 0.004337176156695932, "learning_rate": 2.6933107395204926e-06, "loss": 3.791246490436606e-05, "num_tokens": 982472.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 204, "step_time": 7.7095311099999435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.037428947165608406, "epoch": 1.6015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.005030359141528606, "kl": 0.002402382204309106, "learning_rate": 2.670606033411678e-06, "loss": 2.403911275905557e-05, "num_tokens": 986525.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 205, "step_time": 5.898442960000239 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 23.875, "completions/mean_terminated_length": 23.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06124606728553772, "epoch": 1.609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.017411913722753525, "kl": 0.004249897378031164, "learning_rate": 2.6478871806746496e-06, "loss": 3.866076440317556e-05, "num_tokens": 991340.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 206, "step_time": 7.1736011519999465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.0, "completions/mean_terminated_length": 24.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07086298987269402, "epoch": 1.6171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.007288594264537096, "kl": 0.0015791382757015526, "learning_rate": 2.625156065153473e-06, "loss": 1.550133674754761e-05, "num_tokens": 996212.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 207, "step_time": 7.194994807000057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 18.875, "completions/mean_terminated_length": 18.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10177021846175194, "epoch": 1.625, "frac_reward_zero_std": 1.0, "grad_norm": 0.012765838764607906, "kl": 0.003196087433025241, "learning_rate": 2.602414571709036e-06, "loss": 3.193688462488353e-05, "num_tokens": 1001759.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 208, "step_time": 7.412187181000036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13615204393863678, "epoch": 1.6328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.01755693182349205, "kl": 0.0030227339593693614, "learning_rate": 2.5796645860627665e-06, "loss": 3.0503688321914524e-05, "num_tokens": 1007265.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 209, "step_time": 5.581415800000059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 28.75, "completions/mean_terminated_length": 28.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.15063174068927765, "epoch": 1.640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007047915831208229, "kl": 0.002616984653286636, "learning_rate": 2.556907994640264e-06, "loss": 2.5744397134985775e-05, "num_tokens": 1011331.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 210, "step_time": 7.517502090000107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.375, "completions/mean_terminated_length": 21.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05742892995476723, "epoch": 1.6484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.006198030896484852, "kl": 0.00281632284168154, "learning_rate": 2.5341466844148775e-06, "loss": 2.817007407429628e-05, "num_tokens": 1016182.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 211, "step_time": 6.626822569000069 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 30.875, "completions/mean_terminated_length": 30.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06470214761793613, "epoch": 1.65625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007548889145255089, "kl": 0.0024077071575447917, "learning_rate": 2.511382542751239e-06, "loss": 2.439593299641274e-05, "num_tokens": 1021133.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 212, "step_time": 9.792585082999722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 45.125, "completions/mean_terminated_length": 33.28571701049805, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08313845098018646, "epoch": 1.6640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007846017368137836, "kl": 0.0020210095099173486, "learning_rate": 2.488617457248761e-06, "loss": 1.9834918930428103e-05, "num_tokens": 1026074.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 213, "step_time": 14.740446427000279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03169822786003351, "epoch": 1.671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.004874436184763908, "kl": 0.0018856602837331593, "learning_rate": 2.465853315585123e-06, "loss": 1.8856600945582613e-05, "num_tokens": 1030230.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 214, "step_time": 5.843885092000164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 22.75, "completions/mean_terminated_length": 22.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11479110643267632, "epoch": 1.6796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01489281002432108, "kl": 0.003329504979774356, "learning_rate": 2.443092005359736e-06, "loss": 3.363801079103723e-05, "num_tokens": 1034984.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 215, "step_time": 7.630429413000456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 23.25, "completions/mean_terminated_length": 23.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06059539131820202, "epoch": 1.6875, "frac_reward_zero_std": 1.0, "grad_norm": 0.011219343170523643, "kl": 0.002763925353065133, "learning_rate": 2.420335413937234e-06, "loss": 2.7146501452079974e-05, "num_tokens": 1039826.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 216, "step_time": 7.484693694999805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 24.125, "completions/mean_terminated_length": 24.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06515759788453579, "epoch": 1.6953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.006574144121259451, "kl": 0.0019047525711357594, "learning_rate": 2.3975854282909645e-06, "loss": 1.847416569944471e-05, "num_tokens": 1044679.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 217, "step_time": 7.4298257559999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05187015235424042, "epoch": 1.703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.008155517280101776, "kl": 0.002550492761656642, "learning_rate": 2.374843934846528e-06, "loss": 2.506848977645859e-05, "num_tokens": 1048823.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 218, "step_time": 7.006674997000118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 15.875, "completions/mean_terminated_length": 15.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12386251240968704, "epoch": 1.7109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.019120769575238228, "kl": 0.003044891287572682, "learning_rate": 2.35211281932535e-06, "loss": 3.0685067031299695e-05, "num_tokens": 1054326.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 219, "step_time": 7.376969552999526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.5, "completions/mean_terminated_length": 13.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12342415750026703, "epoch": 1.71875, "frac_reward_zero_std": 1.0, "grad_norm": 0.017717929556965828, "kl": 0.004510085214860737, "learning_rate": 2.3293939665883233e-06, "loss": 4.5417702494887635e-05, "num_tokens": 1059834.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 220, "step_time": 5.555540719999954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 24.875, "completions/mean_terminated_length": 24.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05634631961584091, "epoch": 1.7265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.008950191549956799, "kl": 0.0023297353181988, "learning_rate": 2.306689260479508e-06, "loss": 2.305710586369969e-05, "num_tokens": 1064757.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 221, "step_time": 7.691754480999407 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06376663595438004, "epoch": 1.734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.007804343942552805, "kl": 0.0021846327581442893, "learning_rate": 2.284000583669933e-06, "loss": 2.1814143110532314e-05, "num_tokens": 1069550.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 222, "step_time": 6.525545030000103 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08469109237194061, "epoch": 1.7421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.008673850446939468, "kl": 0.0024135791463777423, "learning_rate": 2.261329817501475e-06, "loss": 2.4780489184195176e-05, "num_tokens": 1074362.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 223, "step_time": 11.37344323599973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07181186601519585, "epoch": 1.75, "frac_reward_zero_std": 1.0, "grad_norm": 0.00881747156381607, "kl": 0.0025651361793279648, "learning_rate": 2.238678841830867e-06, "loss": 2.4443201255053282e-05, "num_tokens": 1079154.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 224, "step_time": 6.7924440969995885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07576226070523262, "epoch": 1.7578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.009993416257202625, "kl": 0.0031048988457769156, "learning_rate": 2.2160495348738127e-06, "loss": 3.1577099434798583e-05, "num_tokens": 1084034.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 225, "step_time": 6.582048697000118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06813675723969936, "epoch": 1.765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.013243633322417736, "kl": 0.0029321471229195595, "learning_rate": 2.1934437730492544e-06, "loss": 2.889215829782188e-05, "num_tokens": 1088838.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 226, "step_time": 6.774607225000182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 27.625, "completions/mean_terminated_length": 27.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0714375339448452, "epoch": 1.7734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.006967134308069944, "kl": 0.0022760473075322807, "learning_rate": 2.1708634308237687e-06, "loss": 2.1496147383004427e-05, "num_tokens": 1093735.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 227, "step_time": 11.807745952999994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11793160066008568, "epoch": 1.78125, "frac_reward_zero_std": 1.0, "grad_norm": 0.05189337208867073, "kl": 0.014766670181415975, "learning_rate": 2.1483103805561493e-06, "loss": 0.00013642504927702248, "num_tokens": 1098522.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 228, "step_time": 6.672160937999706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06688312441110611, "epoch": 1.7890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.011055883020162582, "kl": 0.0036998813739046454, "learning_rate": 2.1257864923421405e-06, "loss": 3.6064928281120956e-05, "num_tokens": 1103364.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 229, "step_time": 6.56204488100002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 33.875, "completions/mean_terminated_length": 33.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.051064563915133476, "epoch": 1.796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.008055821992456913, "kl": 0.0023585216840729117, "learning_rate": 2.1032936338593716e-06, "loss": 2.3143395083025098e-05, "num_tokens": 1107471.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 230, "step_time": 12.61657659399998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 25.875, "completions/mean_terminated_length": 25.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0740746557712555, "epoch": 1.8046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.007505510468035936, "kl": 0.0018645531381480396, "learning_rate": 2.080833670212498e-06, "loss": 1.8865794118028134e-05, "num_tokens": 1112306.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 231, "step_time": 11.113060962999953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0632933359593153, "epoch": 1.8125, "frac_reward_zero_std": 1.0, "grad_norm": 0.005921315401792526, "kl": 0.0018682752852328122, "learning_rate": 2.0584084637785316e-06, "loss": 1.870766755018849e-05, "num_tokens": 1117095.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 232, "step_time": 6.704811484000402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 23.875, "completions/mean_terminated_length": 23.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08165674656629562, "epoch": 1.8203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.012375043705105782, "kl": 0.0033231035922653973, "learning_rate": 2.036019874052428e-06, "loss": 3.117723827017471e-05, "num_tokens": 1121946.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 233, "step_time": 7.801074556999993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08600634709000587, "epoch": 1.828125, "frac_reward_zero_std": 0.5, "grad_norm": 1.3644095659255981, "kl": 0.06359560182318091, "learning_rate": 2.0136697574928853e-06, "loss": 0.06466268002986908, "num_tokens": 1126706.0, "reward": 0.875, "reward_std": 0.3535533845424652, "rewards/reward_fn/mean": 0.875, "rewards/reward_fn/std": 0.3535533845424652, "step": 234, "step_time": 6.896724009000081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08186899498105049, "epoch": 1.8359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.011324622668325901, "kl": 0.003795566619373858, "learning_rate": 1.991359967368416e-06, "loss": 3.484051558189094e-05, "num_tokens": 1131529.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 235, "step_time": 7.659685747999902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.072533143684268, "epoch": 1.84375, "frac_reward_zero_std": 1.0, "grad_norm": 0.015638258308172226, "kl": 0.003511433256790042, "learning_rate": 1.9690923536036673e-06, "loss": 3.5228476917836815e-05, "num_tokens": 1136314.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 236, "step_time": 6.76509731599981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0642678551375866, "epoch": 1.8515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.006433664355427027, "kl": 0.0019541659858077765, "learning_rate": 1.9468687626260314e-06, "loss": 1.9541659639799036e-05, "num_tokens": 1141102.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 237, "step_time": 6.493211311999858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03841168060898781, "epoch": 1.859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.014958206564188004, "kl": 0.004870379460044205, "learning_rate": 1.9246910372125345e-06, "loss": 4.770414670929313e-05, "num_tokens": 1145026.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 238, "step_time": 5.479256311999961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 16.75, "completions/mean_terminated_length": 16.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11552144214510918, "epoch": 1.8671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.06090380623936653, "kl": 0.012604215648025274, "learning_rate": 1.9025610163370385e-06, "loss": 0.0001225811429321766, "num_tokens": 1150584.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 239, "step_time": 7.893678880999687 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06876265630126, "epoch": 1.875, "frac_reward_zero_std": 1.0, "grad_norm": 0.009387739934027195, "kl": 0.003080976544879377, "learning_rate": 1.8804805350177507e-06, "loss": 2.9121343686711043e-05, "num_tokens": 1155364.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 240, "step_time": 6.4674589349997404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.14056620001792908, "epoch": 1.8828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.051601000130176544, "kl": 0.009546731133013964, "learning_rate": 1.8584514241650667e-06, "loss": 9.567832603352144e-05, "num_tokens": 1160847.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 241, "step_time": 5.5828178320002735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.875, "completions/mean_terminated_length": 27.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.036137545481324196, "epoch": 1.890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.009184165857732296, "kl": 0.0031361767323687673, "learning_rate": 1.8364755104297477e-06, "loss": 3.063581243623048e-05, "num_tokens": 1165038.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 242, "step_time": 5.927890447000209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 22.75, "completions/mean_terminated_length": 22.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07793265208601952, "epoch": 1.8984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.02107059955596924, "kl": 0.006186584243550897, "learning_rate": 1.8145546160514622e-06, "loss": 6.0147060139570385e-05, "num_tokens": 1169948.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 243, "step_time": 7.625320269999975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 13.0, "completions/max_terminated_length": 13.0, "completions/mean_length": 13.0, "completions/mean_terminated_length": 13.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12505637109279633, "epoch": 1.90625, "frac_reward_zero_std": 1.0, "grad_norm": 0.020931562408804893, "kl": 0.004113408271223307, "learning_rate": 1.792690558707675e-06, "loss": 4.1134080674964935e-05, "num_tokens": 1175424.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 244, "step_time": 5.521570587000042 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.25, "completions/mean_terminated_length": 21.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.043781496584415436, "epoch": 1.9140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.025867925956845284, "kl": 0.010342990979552269, "learning_rate": 1.7708851513629376e-06, "loss": 9.541432518744841e-05, "num_tokens": 1179346.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 245, "step_time": 5.745181904000219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 20.875, "completions/mean_terminated_length": 20.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07973140105605125, "epoch": 1.921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01369334477931261, "kl": 0.004745342535898089, "learning_rate": 1.7491402021185489e-06, "loss": 4.806690776604228e-05, "num_tokens": 1184229.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 246, "step_time": 6.7438787659998525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 30.625, "completions/mean_terminated_length": 16.71428680419922, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09480598196387291, "epoch": 1.9296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.04195118695497513, "kl": 0.008697986835613847, "learning_rate": 1.7274575140626318e-06, "loss": 7.138060755096376e-05, "num_tokens": 1189874.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 247, "step_time": 15.305213977000221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12500061094760895, "epoch": 1.9375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0684664249420166, "kl": 0.013888376764953136, "learning_rate": 1.7058388851206187e-06, "loss": 0.00013920964556746185, "num_tokens": 1195381.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 248, "step_time": 5.70922280700006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03869021311402321, "epoch": 1.9453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.015580818057060242, "kl": 0.004994981922209263, "learning_rate": 1.6842861079061717e-06, "loss": 4.9939146265387535e-05, "num_tokens": 1199568.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 249, "step_time": 6.0302398560002075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.875, "completions/mean_terminated_length": 27.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03580416925251484, "epoch": 1.953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.012648792937397957, "kl": 0.004584258887916803, "learning_rate": 1.6628009695725348e-06, "loss": 4.5331114961300045e-05, "num_tokens": 1203687.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 250, "step_time": 6.181951604999995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11927280202507973, "epoch": 1.9609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.06169552356004715, "kl": 0.016291253734380007, "learning_rate": 1.6413852516643468e-06, "loss": 0.0001629125326871872, "num_tokens": 1209189.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 251, "step_time": 5.695720361999975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10105976462364197, "epoch": 1.96875, "frac_reward_zero_std": 1.0, "grad_norm": 0.04228847101330757, "kl": 0.010480482131242752, "learning_rate": 1.6200407299699141e-06, "loss": 9.954295819625258e-05, "num_tokens": 1214742.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 252, "step_time": 7.750054464999721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.04526445455849171, "epoch": 1.9765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.03174661472439766, "kl": 0.011470308061689138, "learning_rate": 1.5987691743739636e-06, "loss": 0.00011257198639214039, "num_tokens": 1218788.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 253, "step_time": 6.203933252000752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.125, "completions/mean_terminated_length": 13.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12663870304822922, "epoch": 1.984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.02504121884703636, "kl": 0.005651417304761708, "learning_rate": 1.5775723487108821e-06, "loss": 5.668877565767616e-05, "num_tokens": 1224269.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 254, "step_time": 5.6966200050001135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06413119472563267, "epoch": 1.9921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.036313824355602264, "kl": 0.00865871855057776, "learning_rate": 1.5564520106184643e-06, "loss": 8.569318742956966e-05, "num_tokens": 1229052.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 255, "step_time": 6.825095515999692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 15.0, "completions/mean_terminated_length": 15.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09427187219262123, "epoch": 2.0, "frac_reward_zero_std": 1.0, "grad_norm": 0.028769060969352722, "kl": 0.010795064270496368, "learning_rate": 1.5354099113921614e-06, "loss": 0.0001041956347762607, "num_tokens": 1233740.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 256, "step_time": 6.602496646999953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 22.25, "completions/mean_terminated_length": 22.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0705322865396738, "epoch": 2.0078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.01884527876973152, "kl": 0.004939868813380599, "learning_rate": 1.514447795839874e-06, "loss": 4.6844143071211874e-05, "num_tokens": 1238474.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 257, "step_time": 7.474117505999857 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07266378402709961, "epoch": 2.015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.022645561024546623, "kl": 0.0061811115592718124, "learning_rate": 1.493567402137263e-06, "loss": 6.193597073433921e-05, "num_tokens": 1243295.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 258, "step_time": 6.768364107999787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.0, "completions/mean_terminated_length": 16.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12317134439945221, "epoch": 2.0234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.027730176225304604, "kl": 0.00469887419603765, "learning_rate": 1.4727704616836297e-06, "loss": 4.60884184576571e-05, "num_tokens": 1248799.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 259, "step_time": 7.577763272000084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.25, "completions/mean_terminated_length": 17.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07146281003952026, "epoch": 2.03125, "frac_reward_zero_std": 1.0, "grad_norm": 0.032720040529966354, "kl": 0.007956057321280241, "learning_rate": 1.4520586989583406e-06, "loss": 7.956057379487902e-05, "num_tokens": 1253513.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 260, "step_time": 6.917914829999518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.75, "completions/mean_terminated_length": 21.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06579671613872051, "epoch": 2.0390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.02816769853234291, "kl": 0.008500087773427367, "learning_rate": 1.431433831377836e-06, "loss": 7.657032983843237e-05, "num_tokens": 1258259.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 261, "step_time": 6.894851472000482 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.25, "completions/mean_terminated_length": 13.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13017303496599197, "epoch": 2.046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.027672773227095604, "kl": 0.005289856111630797, "learning_rate": 1.4108975691532273e-06, "loss": 5.289855835144408e-05, "num_tokens": 1263765.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 262, "step_time": 5.753936429000078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07268621399998665, "epoch": 2.0546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.007546247914433479, "kl": 0.002283608540892601, "learning_rate": 1.3904516151484794e-06, "loss": 2.186072015319951e-05, "num_tokens": 1268632.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 263, "step_time": 6.862648165999872 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 16.0, "completions/mean_terminated_length": 16.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11497627571225166, "epoch": 2.0625, "frac_reward_zero_std": 1.0, "grad_norm": 0.013943173922598362, "kl": 0.002583169029094279, "learning_rate": 1.370097664739212e-06, "loss": 2.6390163839096203e-05, "num_tokens": 1274160.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 264, "step_time": 7.761295357000108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.375, "completions/mean_terminated_length": 16.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11228630319237709, "epoch": 2.0703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.02251162752509117, "kl": 0.003555023460648954, "learning_rate": 1.3498374056721198e-06, "loss": 3.469496368779801e-05, "num_tokens": 1279691.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 265, "step_time": 7.840798843000357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.0, "completions/mean_terminated_length": 21.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06668232567608356, "epoch": 2.078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.005440605338662863, "kl": 0.0017036688514053822, "learning_rate": 1.3296725179250274e-06, "loss": 1.7276026483159512e-05, "num_tokens": 1284475.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 266, "step_time": 6.7484774439999455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 18.125, "completions/mean_terminated_length": 18.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08780457451939583, "epoch": 2.0859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.02105136401951313, "kl": 0.006752714281901717, "learning_rate": 1.3096046735675795e-06, "loss": 6.217646296136081e-05, "num_tokens": 1289212.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 267, "step_time": 7.876546445999793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03588521108031273, "epoch": 2.09375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0117345554754138, "kl": 0.003772490192204714, "learning_rate": 1.2896355366226e-06, "loss": 3.7270961911417544e-05, "num_tokens": 1293364.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 268, "step_time": 5.872368308999739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06518647447228432, "epoch": 2.1015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.021700672805309296, "kl": 0.004600343410857022, "learning_rate": 1.2697667629281025e-06, "loss": 4.466335667530075e-05, "num_tokens": 1298152.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 269, "step_time": 6.83284532000016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03484191186726093, "epoch": 2.109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0065823025070130825, "kl": 0.002110914676450193, "learning_rate": 1.2500000000000007e-06, "loss": 2.08322726393817e-05, "num_tokens": 1302192.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 270, "step_time": 5.96173259499983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.25, "completions/mean_terminated_length": 16.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12080564349889755, "epoch": 2.1171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01397998258471489, "kl": 0.0026146003510802984, "learning_rate": 1.2303368868954848e-06, "loss": 2.3293352569453418e-05, "num_tokens": 1307722.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 271, "step_time": 7.747071295000296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 15.125, "completions/mean_terminated_length": 15.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09482500702142715, "epoch": 2.125, "frac_reward_zero_std": 1.0, "grad_norm": 0.014719804748892784, "kl": 0.0029105256544426084, "learning_rate": 1.2107790540771208e-06, "loss": 3.1068855605553836e-05, "num_tokens": 1312491.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 272, "step_time": 6.803867777999585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.0, "completions/mean_terminated_length": 17.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08690010197460651, "epoch": 2.1328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.012722927145659924, "kl": 0.0026088722806889564, "learning_rate": 1.1913281232776445e-06, "loss": 3.0089617212070152e-05, "num_tokens": 1317183.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 273, "step_time": 6.73350407099997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 18.875, "completions/mean_terminated_length": 18.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10723384842276573, "epoch": 2.140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.012960345484316349, "kl": 0.001953385421074927, "learning_rate": 1.1719857073654923e-06, "loss": 1.958140819624532e-05, "num_tokens": 1322710.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 274, "step_time": 7.730757156000436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.03018476441502571, "epoch": 2.1484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0026580565609037876, "kl": 0.0016376653802581131, "learning_rate": 1.1527534102110613e-06, "loss": 1.6376652638427913e-05, "num_tokens": 1326802.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 275, "step_time": 5.899295565000557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.375, "completions/mean_terminated_length": 21.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.059426622465252876, "epoch": 2.15625, "frac_reward_zero_std": 1.0, "grad_norm": 0.025692613795399666, "kl": 0.007004954153671861, "learning_rate": 1.1336328265537195e-06, "loss": 7.027122774161398e-05, "num_tokens": 1331677.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 276, "step_time": 6.831976745000247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 35.375, "completions/mean_terminated_length": 35.375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.03917406965047121, "epoch": 2.1640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0032123220153152943, "kl": 0.001322296040598303, "learning_rate": 1.1146255418695635e-06, "loss": 1.3043942090007477e-05, "num_tokens": 1335820.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 277, "step_time": 10.114057109999976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 24.625, "completions/mean_terminated_length": 24.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.04752049781382084, "epoch": 2.171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.010773031041026115, "kl": 0.004988847824279219, "learning_rate": 1.0957331322399575e-06, "loss": 4.370003443909809e-05, "num_tokens": 1339885.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 278, "step_time": 5.81339948699997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 28.875, "completions/mean_terminated_length": 28.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07032647728919983, "epoch": 2.1796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.005957606248557568, "kl": 0.0015352739137597382, "learning_rate": 1.0769571642208404e-06, "loss": 1.4340959751280025e-05, "num_tokens": 1344764.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 279, "step_time": 11.989179764000255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 20.0, "completions/mean_terminated_length": 20.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.15787944197654724, "epoch": 2.1875, "frac_reward_zero_std": 1.0, "grad_norm": 0.022516168653964996, "kl": 0.006525776581838727, "learning_rate": 1.0582991947128324e-06, "loss": 6.593053694814444e-05, "num_tokens": 1348672.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 280, "step_time": 6.449207413999829 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 17.25, "completions/mean_terminated_length": 17.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0816731434315443, "epoch": 2.1953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.010547587648034096, "kl": 0.002592157106846571, "learning_rate": 1.0397607708321302e-06, "loss": 2.5727105821715668e-05, "num_tokens": 1353414.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 281, "step_time": 7.077126720000251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.033613706938922405, "epoch": 2.203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0056767649948596954, "kl": 0.0027879257686436176, "learning_rate": 1.0213434297822275e-06, "loss": 2.6398176487418823e-05, "num_tokens": 1357505.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 282, "step_time": 5.923250760999508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 26.875, "completions/mean_terminated_length": 26.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03769804444164038, "epoch": 2.2109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.005933691281825304, "kl": 0.00262093311175704, "learning_rate": 1.0030486987264436e-06, "loss": 2.5216926587745547e-05, "num_tokens": 1361644.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 283, "step_time": 5.9830027580001115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 17.875, "completions/mean_terminated_length": 17.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0936516635119915, "epoch": 2.21875, "frac_reward_zero_std": 1.0, "grad_norm": 0.015039868652820587, "kl": 0.003393694059923291, "learning_rate": 9.848780946612962e-07, "loss": 3.5618319088825956e-05, "num_tokens": 1366347.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 284, "step_time": 7.200875815000018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06426311284303665, "epoch": 2.2265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007608731277287006, "kl": 0.001917863031849265, "learning_rate": 9.66833124290709e-07, "loss": 1.9463377611828037e-05, "num_tokens": 1371241.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 285, "step_time": 6.8967751259997385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 27.25, "completions/mean_terminated_length": 27.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06446886621415615, "epoch": 2.234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.010835153982043266, "kl": 0.00325047189835459, "learning_rate": 9.489152839010799e-07, "loss": 2.8311722417129204e-05, "num_tokens": 1376131.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 286, "step_time": 11.860804265999832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 17.5, "completions/mean_terminated_length": 17.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08248795196413994, "epoch": 2.2421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.010750790126621723, "kl": 0.0020258878357708454, "learning_rate": 9.311260592372045e-07, "loss": 2.2169147996464744e-05, "num_tokens": 1380907.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 287, "step_time": 6.986179221999919 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11641774699091911, "epoch": 2.25, "frac_reward_zero_std": 1.0, "grad_norm": 0.011737891472876072, "kl": 0.003091278951615095, "learning_rate": 9.134669253790814e-07, "loss": 2.849579141184222e-05, "num_tokens": 1385623.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 288, "step_time": 6.810695373999806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.125, "completions/mean_terminated_length": 21.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06711885333061218, "epoch": 2.2578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.007158924825489521, "kl": 0.0025333911762572825, "learning_rate": 8.959393466195973e-07, "loss": 2.5275945517932996e-05, "num_tokens": 1390344.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 289, "step_time": 6.85957102600014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 17.25, "completions/mean_terminated_length": 17.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07853703573346138, "epoch": 2.265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.01479937881231308, "kl": 0.004124688799493015, "learning_rate": 8.785447763431101e-07, "loss": 4.1246887121815234e-05, "num_tokens": 1395054.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 290, "step_time": 6.63891823799986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 128.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 40.75, "completions/mean_terminated_length": 28.285715103149414, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0656740702688694, "epoch": 2.2734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0060638003051280975, "kl": 0.0017024054541252553, "learning_rate": 8.612846569049324e-07, "loss": 1.726844857330434e-05, "num_tokens": 1400052.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 291, "step_time": 14.556338565999795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 23.375, "completions/mean_terminated_length": 23.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03545568138360977, "epoch": 2.28125, "frac_reward_zero_std": 1.0, "grad_norm": 0.007708441931754351, "kl": 0.0022156074410304427, "learning_rate": 8.441604195117315e-07, "loss": 2.2025331418262795e-05, "num_tokens": 1404051.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 292, "step_time": 5.789175566999802 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.09139403142035007, "epoch": 2.2890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007536021526902914, "kl": 0.0019620120874606073, "learning_rate": 8.271734841028553e-07, "loss": 1.879773844848387e-05, "num_tokens": 1408800.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 293, "step_time": 6.592784782999843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.5, "completions/mean_terminated_length": 21.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03953526355326176, "epoch": 2.296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.01195154432207346, "kl": 0.003599834628403187, "learning_rate": 8.103252592325897e-07, "loss": 3.324213685118593e-05, "num_tokens": 1412752.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 294, "step_time": 5.7077796439998565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.125, "completions/mean_terminated_length": 17.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08125988394021988, "epoch": 2.3046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.007471000775694847, "kl": 0.00221208727452904, "learning_rate": 7.936171419533653e-07, "loss": 2.2959266061661765e-05, "num_tokens": 1417441.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 295, "step_time": 6.576927980000164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1136050745844841, "epoch": 2.3125, "frac_reward_zero_std": 1.0, "grad_norm": 0.011644271202385426, "kl": 0.0031743868021294475, "learning_rate": 7.770505176999066e-07, "loss": 2.477930684108287e-05, "num_tokens": 1422995.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 296, "step_time": 7.677933532000225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07930410280823708, "epoch": 2.3203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.006496694404631853, "kl": 0.0013743448653258383, "learning_rate": 7.606267601743614e-07, "loss": 1.3506290997611359e-05, "num_tokens": 1427804.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 297, "step_time": 6.850105020000228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1292087510228157, "epoch": 2.328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.02982700616121292, "kl": 0.002042026782874018, "learning_rate": 7.443472312323824e-07, "loss": 1.8672115402296185e-05, "num_tokens": 1433332.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 298, "step_time": 7.489191680000204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.375, "completions/mean_terminated_length": 21.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05677143856883049, "epoch": 2.3359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.007880290038883686, "kl": 0.0026660520816221833, "learning_rate": 7.282132807702144e-07, "loss": 2.6689433070714585e-05, "num_tokens": 1438211.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 299, "step_time": 6.8618144679999205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0277756555005908, "epoch": 2.34375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0013915153685957193, "kl": 0.001870743464678526, "learning_rate": 7.122262466127513e-07, "loss": 1.8572343833511695e-05, "num_tokens": 1442435.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 300, "step_time": 6.021347086999867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05828935466706753, "epoch": 2.3515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.004984916187822819, "kl": 0.0012245121179148555, "learning_rate": 6.963874544026109e-07, "loss": 1.2662912922678515e-05, "num_tokens": 1447379.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 301, "step_time": 13.41408500999978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.25, "completions/mean_terminated_length": 16.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12101850658655167, "epoch": 2.359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.06349971890449524, "kl": 0.013034343253821135, "learning_rate": 6.806982174902065e-07, "loss": 0.00013196206418797374, "num_tokens": 1452909.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 302, "step_time": 7.468966810999973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 29.5, "completions/mean_terminated_length": 29.5, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.02912551909685135, "epoch": 2.3671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0009350188192911446, "kl": 0.0012789619504474103, "learning_rate": 6.651598368248494e-07, "loss": 1.2786756997229531e-05, "num_tokens": 1456989.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 303, "step_time": 5.737073586000406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 28.875, "completions/mean_terminated_length": 28.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08968518674373627, "epoch": 2.375, "frac_reward_zero_std": 1.0, "grad_norm": 0.010912664234638214, "kl": 0.002259014407172799, "learning_rate": 6.497736008468703e-07, "loss": 2.482989293639548e-05, "num_tokens": 1461908.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 304, "step_time": 13.122933298000135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06897314265370369, "epoch": 2.3828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0196547731757164, "kl": 0.006106121756602079, "learning_rate": 6.345407853807864e-07, "loss": 5.6373894040007144e-05, "num_tokens": 1466729.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 305, "step_time": 7.000840238999899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.625, "completions/mean_terminated_length": 13.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13361556828022003, "epoch": 2.390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.02360568754374981, "kl": 0.0038382920320145786, "learning_rate": 6.194626535295059e-07, "loss": 3.8651305658277124e-05, "num_tokens": 1472214.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 306, "step_time": 5.796944548999818 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07942482270300388, "epoch": 2.3984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.004173581022769213, "kl": 0.0015490494552068412, "learning_rate": 6.045404555695935e-07, "loss": 1.529452310933266e-05, "num_tokens": 1476946.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 307, "step_time": 6.881738667000263 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 16.625, "completions/mean_terminated_length": 16.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11725720018148422, "epoch": 2.40625, "frac_reward_zero_std": 1.0, "grad_norm": 0.027200696989893913, "kl": 0.006251339975278825, "learning_rate": 5.897754288475979e-07, "loss": 5.222540858085267e-05, "num_tokens": 1482503.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 308, "step_time": 7.865210730999479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.375, "completions/mean_terminated_length": 27.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03089694306254387, "epoch": 2.4140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.001481092651374638, "kl": 0.0015604888903908432, "learning_rate": 5.751687976774523e-07, "loss": 1.5621018974343315e-05, "num_tokens": 1486670.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 309, "step_time": 5.85635403499964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.25, "completions/mean_terminated_length": 21.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07554454170167446, "epoch": 2.421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.005015636794269085, "kl": 0.001698852691333741, "learning_rate": 5.607217732389503e-07, "loss": 1.6451504052383825e-05, "num_tokens": 1491408.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 310, "step_time": 6.528806915999667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 21.75, "completions/mean_terminated_length": 21.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12106388807296753, "epoch": 2.4296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.02396441251039505, "kl": 0.0038617003010585904, "learning_rate": 5.464355534773217e-07, "loss": 2.8673672204604372e-05, "num_tokens": 1496958.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 311, "step_time": 7.533618949999891 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 18.875, "completions/mean_terminated_length": 18.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11433938518166542, "epoch": 2.4375, "frac_reward_zero_std": 1.0, "grad_norm": 0.003641946241259575, "kl": 0.001054832770023495, "learning_rate": 5.323113230038899e-07, "loss": 9.623323421692476e-06, "num_tokens": 1502481.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 312, "step_time": 7.2610713440003565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0287599079310894, "epoch": 2.4453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.002698131836950779, "kl": 0.0020212933886796236, "learning_rate": 5.183502529978548e-07, "loss": 2.0212932213325985e-05, "num_tokens": 1506701.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 313, "step_time": 5.806462265000391 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.375, "completions/mean_terminated_length": 25.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.032324470579624176, "epoch": 2.453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0024203648790717125, "kl": 0.0017645764746703207, "learning_rate": 5.045535011091693e-07, "loss": 1.707239425741136e-05, "num_tokens": 1510856.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 314, "step_time": 5.994479979000062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 17.25, "completions/mean_terminated_length": 17.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07266517356038094, "epoch": 2.4609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.009694607928395271, "kl": 0.0022006782237440348, "learning_rate": 4.909222113625545e-07, "loss": 2.202560062869452e-05, "num_tokens": 1515614.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 315, "step_time": 6.553128839000237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.25, "completions/mean_terminated_length": 16.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1260695531964302, "epoch": 2.46875, "frac_reward_zero_std": 1.0, "grad_norm": 0.02785095013678074, "kl": 0.003913273336365819, "learning_rate": 4.774575140626317e-07, "loss": 3.9146947528934106e-05, "num_tokens": 1521120.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 316, "step_time": 7.450610239000071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.02743313182145357, "epoch": 2.4765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0003154528676532209, "kl": 0.0018573110573925078, "learning_rate": 4.6416052570020047e-07, "loss": 1.8573109628050588e-05, "num_tokens": 1525348.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 317, "step_time": 6.067128046000107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 27.5, "completions/mean_terminated_length": 27.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03275044448673725, "epoch": 2.484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0030340480152517557, "kl": 0.0018103363690897822, "learning_rate": 4.510323488596588e-07, "loss": 1.801705002435483e-05, "num_tokens": 1529396.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 318, "step_time": 6.102241871999922 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 31.25, "completions/mean_terminated_length": 31.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.044731512665748596, "epoch": 2.4921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.006392910145223141, "kl": 0.002438001334667206, "learning_rate": 4.380740721275786e-07, "loss": 2.4828672394505702e-05, "num_tokens": 1533658.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 319, "step_time": 8.285051075999945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11057645827531815, "epoch": 2.5, "frac_reward_zero_std": 1.0, "grad_norm": 0.03642839938402176, "kl": 0.0035907960700569674, "learning_rate": 4.252867700024374e-07, "loss": 4.713074667961337e-05, "num_tokens": 1539236.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 320, "step_time": 7.745400238999991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.875, "completions/mean_terminated_length": 19.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06698767095804214, "epoch": 2.5078125, "frac_reward_zero_std": 1.0, "grad_norm": 0.007277606055140495, "kl": 0.0023683570325374603, "learning_rate": 4.1267150280552256e-07, "loss": 2.3156695533543825e-05, "num_tokens": 1543991.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 321, "step_time": 7.041696197999954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.625, "completions/mean_terminated_length": 13.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12220295518636703, "epoch": 2.515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.021708490327000618, "kl": 0.005996785359457135, "learning_rate": 4.002293165930088e-07, "loss": 5.987433178233914e-05, "num_tokens": 1549476.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 322, "step_time": 5.8188227460000235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 21.375, "completions/mean_terminated_length": 21.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.060368530452251434, "epoch": 2.5234375, "frac_reward_zero_std": 1.0, "grad_norm": 0.03838368505239487, "kl": 0.009895860450342298, "learning_rate": 3.879612430692223e-07, "loss": 8.399881335208192e-05, "num_tokens": 1554271.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 323, "step_time": 6.701541300000372 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.375, "completions/mean_terminated_length": 19.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07650637812912464, "epoch": 2.53125, "frac_reward_zero_std": 1.0, "grad_norm": 0.004762283060699701, "kl": 0.0014340974448714405, "learning_rate": 3.7586829950108787e-07, "loss": 1.502779468864901e-05, "num_tokens": 1558994.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 324, "step_time": 6.7548362940001425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 29.375, "completions/mean_terminated_length": 29.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.2658330798149109, "epoch": 2.5390625, "frac_reward_zero_std": 1.0, "grad_norm": 0.003241270314902067, "kl": 0.0007280520221684128, "learning_rate": 3.639514886337786e-07, "loss": 8.224497832998168e-06, "num_tokens": 1564653.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 325, "step_time": 10.17069353899933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07305469736456871, "epoch": 2.546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0017727442318573594, "kl": 0.0009516599820926785, "learning_rate": 3.5221179860757156e-07, "loss": 9.93437697616173e-06, "num_tokens": 1569417.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 326, "step_time": 6.583777612000176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03469938971102238, "epoch": 2.5546875, "frac_reward_zero_std": 1.0, "grad_norm": 0.008155560120940208, "kl": 0.0023509636521339417, "learning_rate": 3.4065020287590456e-07, "loss": 2.3022465029498562e-05, "num_tokens": 1573341.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 327, "step_time": 5.557240961000389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.029884813353419304, "epoch": 2.5625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0019974284805357456, "kl": 0.0016899254987947643, "learning_rate": 3.292676601246661e-07, "loss": 1.6899255570024252e-05, "num_tokens": 1577501.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 328, "step_time": 5.841788397000528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 23.25, "completions/mean_terminated_length": 23.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03349179029464722, "epoch": 2.5703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.006807905156165361, "kl": 0.002164472476579249, "learning_rate": 3.18065114192693e-07, "loss": 2.147164923371747e-05, "num_tokens": 1581531.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 329, "step_time": 5.71764247999954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 20.75, "completions/mean_terminated_length": 20.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10336236655712128, "epoch": 2.578125, "frac_reward_zero_std": 1.0, "grad_norm": 0.013856982812285423, "kl": 0.003586582955904305, "learning_rate": 3.0704349399351437e-07, "loss": 3.917415233445354e-05, "num_tokens": 1586265.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 330, "step_time": 6.691765174000011 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0626723226159811, "epoch": 2.5859375, "frac_reward_zero_std": 1.0, "grad_norm": 0.004872177727520466, "kl": 0.0017109083128161728, "learning_rate": 2.962037134383211e-07, "loss": 1.6971382137853652e-05, "num_tokens": 1591170.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 331, "step_time": 7.28137495899955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1446835845708847, "epoch": 2.59375, "frac_reward_zero_std": 1.0, "grad_norm": 0.03924147039651871, "kl": 0.0011937321105506271, "learning_rate": 2.855466713601868e-07, "loss": 1.1964344594161958e-05, "num_tokens": 1596649.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 332, "step_time": 5.465260511999986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 25.25, "completions/mean_terminated_length": 25.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05687732808291912, "epoch": 2.6015625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0024226743262261152, "kl": 0.001208013971336186, "learning_rate": 2.750732514395363e-07, "loss": 1.1762923350033816e-05, "num_tokens": 1601571.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 333, "step_time": 7.622067566999704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.5, "completions/mean_terminated_length": 19.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0665333904325962, "epoch": 2.609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.009631386026740074, "kl": 0.002712785266339779, "learning_rate": 2.647843221308721e-07, "loss": 2.844407754309941e-05, "num_tokens": 1606451.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 334, "step_time": 6.9531723879999845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 23.875, "completions/mean_terminated_length": 23.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.371206421405077, "epoch": 2.6171875, "frac_reward_zero_std": 1.0, "grad_norm": 0.00424219761043787, "kl": 0.0011196635314263403, "learning_rate": 2.5468073659076e-07, "loss": 1.2226804756210186e-05, "num_tokens": 1612018.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 335, "step_time": 10.750021371999992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06773288920521736, "epoch": 2.625, "frac_reward_zero_std": 1.0, "grad_norm": 0.001815224764868617, "kl": 0.001531377958599478, "learning_rate": 2.44763332607087e-07, "loss": 1.508441346231848e-05, "num_tokens": 1616870.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 336, "step_time": 7.147886858999755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 27.25, "completions/mean_terminated_length": 27.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05348302982747555, "epoch": 2.6328125, "frac_reward_zero_std": 1.0, "grad_norm": 0.002274098340421915, "kl": 0.0012252243468537927, "learning_rate": 2.3503293252959136e-07, "loss": 1.2252243323018774e-05, "num_tokens": 1621724.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 337, "step_time": 7.523159131000284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 23.625, "completions/mean_terminated_length": 23.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1177109070122242, "epoch": 2.640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.011636418290436268, "kl": 0.003375065280124545, "learning_rate": 2.2549034320167501e-07, "loss": 3.056726563954726e-05, "num_tokens": 1626517.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 338, "step_time": 7.475345586000003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07587442919611931, "epoch": 2.6484375, "frac_reward_zero_std": 1.0, "grad_norm": 0.003583204001188278, "kl": 0.001523865619674325, "learning_rate": 2.1613635589349756e-07, "loss": 1.486748533352511e-05, "num_tokens": 1631389.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 339, "step_time": 6.883027566999772 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.02992851845920086, "epoch": 2.65625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0010637313826009631, "kl": 0.0017816239269450307, "learning_rate": 2.0697174623636795e-07, "loss": 1.7696058421279304e-05, "num_tokens": 1635621.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 340, "step_time": 5.765514154000357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 24.375, "completions/mean_terminated_length": 24.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06319654919207096, "epoch": 2.6640625, "frac_reward_zero_std": 1.0, "grad_norm": 0.004237866494804621, "kl": 0.0019926169188693166, "learning_rate": 1.9799727415842323e-07, "loss": 2.3374921511276625e-05, "num_tokens": 1640408.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 341, "step_time": 7.587094982999588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12324276193976402, "epoch": 2.671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.004930234979838133, "kl": 0.0005672876577591524, "learning_rate": 1.8921368382162352e-07, "loss": 6.185021447890904e-06, "num_tokens": 1645961.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 342, "step_time": 7.826221096000154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 115.0, "completions/max_terminated_length": 115.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0991355199366808, "epoch": 2.6796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0027120131999254227, "kl": 0.0009974082058761269, "learning_rate": 1.8062170356003854e-07, "loss": 9.134013453149237e-06, "num_tokens": 1650807.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 343, "step_time": 13.32459857799995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.875, "completions/mean_terminated_length": 21.875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07806593738496304, "epoch": 2.6875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0022448606323450804, "kl": 0.0018360121175646782, "learning_rate": 1.7222204581946038e-07, "loss": 1.7249569282284938e-05, "num_tokens": 1655702.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 344, "step_time": 6.476274790999923 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 22.0, "completions/mean_terminated_length": 22.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.0640562754124403, "epoch": 2.6953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.013984336517751217, "kl": 0.002910680166678503, "learning_rate": 1.6401540709832242e-07, "loss": 3.365988959558308e-05, "num_tokens": 1660514.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 345, "step_time": 7.252558954999586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 33.625, "completions/mean_terminated_length": 33.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03660286124795675, "epoch": 2.703125, "frac_reward_zero_std": 1.0, "grad_norm": 0.003199823433533311, "kl": 0.0017720660544000566, "learning_rate": 1.5600246788994938e-07, "loss": 1.750032060954254e-05, "num_tokens": 1664703.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 346, "step_time": 9.787063931000375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 19.625, "completions/mean_terminated_length": 19.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06424449197947979, "epoch": 2.7109375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0034813072998076677, "kl": 0.0016274115769192576, "learning_rate": 1.4818389262612948e-07, "loss": 1.6602698451606557e-05, "num_tokens": 1669584.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 347, "step_time": 6.833064813999499 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 27.25, "completions/mean_terminated_length": 27.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06542891636490822, "epoch": 2.71875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0009135848376899958, "kl": 0.0008284190844278783, "learning_rate": 1.4056032962202038e-07, "loss": 8.49508069222793e-06, "num_tokens": 1674430.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 348, "step_time": 7.179126660000293 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.625, "completions/mean_terminated_length": 13.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1259615123271942, "epoch": 2.7265625, "frac_reward_zero_std": 1.0, "grad_norm": 0.006351651158183813, "kl": 0.0011312126298435032, "learning_rate": 1.3313241102239056e-07, "loss": 1.1335807357681915e-05, "num_tokens": 1679939.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 349, "step_time": 5.610404259999996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.032224283553659916, "epoch": 2.734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0039573912508785725, "kl": 0.0017619336140342057, "learning_rate": 1.2590075274920206e-07, "loss": 1.7461547031416558e-05, "num_tokens": 1684007.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 350, "step_time": 5.557787984000242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.5, "completions/mean_terminated_length": 21.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06698081642389297, "epoch": 2.7421875, "frac_reward_zero_std": 1.0, "grad_norm": 0.002890112577006221, "kl": 0.0011417069763410836, "learning_rate": 1.1886595445053745e-07, "loss": 1.179358150693588e-05, "num_tokens": 1688915.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 351, "step_time": 6.857757833000051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 21.625, "completions/mean_terminated_length": 21.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.05869276635348797, "epoch": 2.75, "frac_reward_zero_std": 1.0, "grad_norm": 0.0013901939382776618, "kl": 0.0012589980033226311, "learning_rate": 1.120285994508799e-07, "loss": 1.2590622645802796e-05, "num_tokens": 1693788.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 352, "step_time": 6.611652337999658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 20.375, "completions/mean_terminated_length": 20.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07492958009243011, "epoch": 2.7578125, "frac_reward_zero_std": 0.5, "grad_norm": 2.069866895675659, "kl": 0.37729693634901196, "learning_rate": 1.053892547027402e-07, "loss": -0.09150571376085281, "num_tokens": 1698687.0, "reward": 0.887499988079071, "reward_std": 0.3181980550289154, "rewards/reward_fn/mean": 0.887499988079071, "rewards/reward_fn/std": 0.3181980550289154, "step": 353, "step_time": 7.622229282999797 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.125, "completions/mean_terminated_length": 19.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07208308577537537, "epoch": 2.765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.001533597824163735, "kl": 0.0011728731915354729, "learning_rate": 9.894847073964875e-08, "loss": 1.1982096111751162e-05, "num_tokens": 1703472.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 354, "step_time": 6.501406307999787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.375, "completions/mean_terminated_length": 13.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13054991513490677, "epoch": 2.7734375, "frac_reward_zero_std": 1.0, "grad_norm": 0.008911830373108387, "kl": 0.0018170861876569688, "learning_rate": 9.270678163050218e-08, "loss": 1.823622551455628e-05, "num_tokens": 1708979.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 355, "step_time": 5.554297769000186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.0, "completions/mean_terminated_length": 23.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.036714909598231316, "epoch": 2.78125, "frac_reward_zero_std": 1.0, "grad_norm": 0.006838838569819927, "kl": 0.002757440961431712, "learning_rate": 8.666470493528007e-08, "loss": 2.4345088604604825e-05, "num_tokens": 1712979.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 356, "step_time": 5.553534669000328 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12004699558019638, "epoch": 2.7890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.003003190504387021, "kl": 0.0006363969732774422, "learning_rate": 8.082274166213016e-08, "loss": 6.007579941069707e-06, "num_tokens": 1718508.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 357, "step_time": 7.342511635000392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.031998704187572, "epoch": 2.796875, "frac_reward_zero_std": 1.0, "grad_norm": 0.0018181306077167392, "kl": 0.0013061033096164465, "learning_rate": 7.518137622582189e-08, "loss": 1.2957580111105926e-05, "num_tokens": 1722584.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 358, "step_time": 5.5584459810002045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 25.0, "completions/mean_terminated_length": 25.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03392230533063412, "epoch": 2.8046875, "frac_reward_zero_std": 1.0, "grad_norm": 0.005436773411929607, "kl": 0.002172080159652978, "learning_rate": 6.974107640758176e-08, "loss": 2.063044303213246e-05, "num_tokens": 1726536.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 359, "step_time": 5.491190248999828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12706465646624565, "epoch": 2.8125, "frac_reward_zero_std": 1.0, "grad_norm": 0.005568970926105976, "kl": 0.0011207733186893165, "learning_rate": 6.450229331630253e-08, "loss": 1.0281852155458182e-05, "num_tokens": 1732041.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 360, "step_time": 7.323580845000379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.028290102258324623, "epoch": 2.8203125, "frac_reward_zero_std": 1.0, "grad_norm": 0.0013824108755216002, "kl": 0.0015246181283146143, "learning_rate": 5.946546135113862e-08, "loss": 1.5194093066384085e-05, "num_tokens": 1736217.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 361, "step_time": 5.700615629000367 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 26.75, "completions/mean_terminated_length": 26.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06074472516775131, "epoch": 2.828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.003173437202349305, "kl": 0.0010228622995782644, "learning_rate": 5.463099816548578e-08, "loss": 1.052904008247424e-05, "num_tokens": 1741115.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 362, "step_time": 7.050874709999789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 20.375, "completions/mean_terminated_length": 20.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07005861215293407, "epoch": 2.8359375, "frac_reward_zero_std": 1.0, "grad_norm": 0.0047007217071950436, "kl": 0.001660762238316238, "learning_rate": 4.999930463234964e-08, "loss": 1.6024239812395535e-05, "num_tokens": 1746014.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 363, "step_time": 7.63879414999974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12410368397831917, "epoch": 2.84375, "frac_reward_zero_std": 1.0, "grad_norm": 0.011131868697702885, "kl": 0.0020194172975607216, "learning_rate": 4.557076481110367e-08, "loss": 2.2339860151987523e-05, "num_tokens": 1751543.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 364, "step_time": 7.354793092999898 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.125, "completions/mean_terminated_length": 13.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.1403006836771965, "epoch": 2.8515625, "frac_reward_zero_std": 1.0, "grad_norm": 0.00545878428965807, "kl": 0.0008273630810435861, "learning_rate": 4.134574591564494e-08, "loss": 8.271908882306889e-06, "num_tokens": 1757048.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 365, "step_time": 5.55601100199965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 23.125, "completions/mean_terminated_length": 23.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07144136913120747, "epoch": 2.859375, "frac_reward_zero_std": 0.5, "grad_norm": 0.8153831958770752, "kl": 0.6430793326580897, "learning_rate": 3.732459828394402e-08, "loss": -0.04161795973777771, "num_tokens": 1761901.0, "reward": 0.8812500238418579, "reward_std": 0.3358757197856903, "rewards/reward_fn/mean": 0.8812500238418579, "rewards/reward_fn/std": 0.3358757197856903, "step": 366, "step_time": 7.062660026999765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06979860179126263, "epoch": 2.8671875, "frac_reward_zero_std": 1.0, "grad_norm": 0.004871731158345938, "kl": 0.0016156822093762457, "learning_rate": 3.3507655348995194e-08, "loss": 1.5751948012621142e-05, "num_tokens": 1766674.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 367, "step_time": 7.081914410000536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.10976235195994377, "epoch": 2.875, "frac_reward_zero_std": 1.0, "grad_norm": 0.004110577516257763, "kl": 0.0005710943078156561, "learning_rate": 2.98952336111677e-08, "loss": 6.769578249077313e-06, "num_tokens": 1772252.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 368, "step_time": 7.688176613999985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 34.625, "completions/mean_terminated_length": 34.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.04085162188857794, "epoch": 2.8828125, "frac_reward_zero_std": 1.0, "grad_norm": 0.002521694405004382, "kl": 0.0012893079547211528, "learning_rate": 2.6487632611962578e-08, "loss": 1.2892131053376943e-05, "num_tokens": 1776517.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 369, "step_time": 9.961404026999844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.08238299190998077, "epoch": 2.890625, "frac_reward_zero_std": 1.0, "grad_norm": 0.007919838652014732, "kl": 0.002668961475137621, "learning_rate": 2.3285134909173113e-08, "loss": 2.5326695322291926e-05, "num_tokens": 1781345.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 370, "step_time": 6.4334861359998285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 16.5, "completions/mean_terminated_length": 16.5, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12997304648160934, "epoch": 2.8984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.00875998567789793, "kl": 0.0007102387316990644, "learning_rate": 2.028800605345771e-08, "loss": 6.7351170400797855e-06, "num_tokens": 1786897.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 371, "step_time": 7.530595849999827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07069241255521774, "epoch": 2.90625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0031238263472914696, "kl": 0.0017178311245515943, "learning_rate": 1.7496494566317247e-08, "loss": 1.785397034836933e-05, "num_tokens": 1791725.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 372, "step_time": 6.554121677999774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.11858610808849335, "epoch": 2.9140625, "frac_reward_zero_std": 1.0, "grad_norm": 0.01251543965190649, "kl": 0.0027800038806162775, "learning_rate": 1.4910831919490997e-08, "loss": 2.9193033697083592e-05, "num_tokens": 1797254.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 373, "step_time": 7.373935341000106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 16.375, "completions/mean_terminated_length": 16.375, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12266703695058823, "epoch": 2.921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.04878854751586914, "kl": 0.011298867233563215, "learning_rate": 1.2531232515760328e-08, "loss": 9.663843229645863e-05, "num_tokens": 1802761.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 374, "step_time": 7.402829858999667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 25.75, "completions/mean_terminated_length": 25.75, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.03546947054564953, "epoch": 2.9296875, "frac_reward_zero_std": 1.0, "grad_norm": 0.006199007388204336, "kl": 0.001919182192068547, "learning_rate": 1.0357893671171793e-08, "loss": 1.9191820683772676e-05, "num_tokens": 1806727.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 375, "step_time": 5.591665662999731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 27.0, "completions/mean_terminated_length": 27.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.030339499935507774, "epoch": 2.9375, "frac_reward_zero_std": 1.0, "grad_norm": 0.001278455718420446, "kl": 0.001506837084889412, "learning_rate": 8.390995598676067e-09, "loss": 1.4967106835683808e-05, "num_tokens": 1810895.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 376, "step_time": 5.731536610000148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 35.5, "completions/mean_terminated_length": 35.5, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.03009831439703703, "epoch": 2.9453125, "frac_reward_zero_std": 1.0, "grad_norm": 0.006957578472793102, "kl": 0.001761360326781869, "learning_rate": 6.63070139318378e-09, "loss": 1.8286591512151062e-05, "num_tokens": 1815119.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 377, "step_time": 9.591905867999685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 19.0, "completions/mean_terminated_length": 19.0, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.07046742364764214, "epoch": 2.953125, "frac_reward_zero_std": 1.0, "grad_norm": 0.007482482586055994, "kl": 0.0023475169437006116, "learning_rate": 5.077157018041623e-09, "loss": 2.212448998761829e-05, "num_tokens": 1819847.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 378, "step_time": 6.737704846999804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 14.0, "completions/mean_terminated_length": 14.0, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.12937615811824799, "epoch": 2.9609375, "frac_reward_zero_std": 1.0, "grad_norm": 0.023704534396529198, "kl": 0.004825950600206852, "learning_rate": 3.730491292930072e-09, "loss": 4.825950600206852e-05, "num_tokens": 1825355.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 379, "step_time": 5.74336707700013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 14.0, "completions/max_terminated_length": 14.0, "completions/mean_length": 13.125, "completions/mean_terminated_length": 13.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.13150636106729507, "epoch": 2.96875, "frac_reward_zero_std": 1.0, "grad_norm": 0.004724206868559122, "kl": 0.0005312660941854119, "learning_rate": 2.590815883181108e-09, "loss": 5.310364940669388e-06, "num_tokens": 1830860.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 380, "step_time": 5.56844295000019 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 16.125, "completions/mean_terminated_length": 16.125, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.12551702931523323, "epoch": 2.9765625, "frac_reward_zero_std": 1.0, "grad_norm": 0.0012423633597791195, "kl": 0.00047561304381815717, "learning_rate": 1.6582252905186779e-09, "loss": 5.338163646229077e-06, "num_tokens": 1836389.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 381, "step_time": 7.517625181999847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 25.625, "completions/mean_terminated_length": 25.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06911934912204742, "epoch": 2.984375, "frac_reward_zero_std": 1.0, "grad_norm": 0.006786187645047903, "kl": 0.0018059620633721352, "learning_rate": 9.32796845223294e-10, "loss": 1.838826938183047e-05, "num_tokens": 1841294.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 382, "step_time": 7.187780472999748 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 28.625, "completions/mean_terminated_length": 28.625, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.06470359116792679, "epoch": 2.9921875, "frac_reward_zero_std": 1.0, "grad_norm": 0.00552311772480607, "kl": 0.0018424552981741726, "learning_rate": 4.1459069971938604e-10, "loss": 1.6110756405396387e-05, "num_tokens": 1846227.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 383, "step_time": 12.270986474999972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.028123882599174976, "epoch": 3.0, "frac_reward_zero_std": 1.0, "grad_norm": 0.0005120371351949871, "kl": 0.0014926688745617867, "learning_rate": 1.0364982358707087e-10, "loss": 1.4926688891137019e-05, "num_tokens": 1850395.0, "reward": 1.0, "reward_std": 0.0, "rewards/reward_fn/mean": 1.0, "rewards/reward_fn/std": 0.0, "step": 384, "step_time": 5.804545400000279 } ], "logging_steps": 1, "max_steps": 384, "num_input_tokens_seen": 1850395, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }